{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,29]],"date-time":"2026-07-29T06:04:30Z","timestamp":1785305070841,"version":"3.55.0"},"reference-count":64,"publisher":"Springer Science and Business Media LLC","issue":"7","license":[{"start":{"date-parts":[[2026,6,20]],"date-time":"2026-06-20T00:00:00Z","timestamp":1781913600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,6,20]],"date-time":"2026-06-20T00:00:00Z","timestamp":1781913600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2026,7]]},"DOI":"10.1007\/s11263-026-02887-z","type":"journal-article","created":{"date-parts":[[2026,6,20]],"date-time":"2026-06-20T05:23:36Z","timestamp":1781933016000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Learning Pseudo 3D Representation for Ego-centric 2D Multiple Object Tracking"],"prefix":"10.1007","volume":"134","author":[{"given":"Jiawei","family":"He","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lue","family":"Fan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Zhaoxiang","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,6,20]]},"reference":[{"key":"2887_CR1","doi-asserted-by":"crossref","unstructured":"Wojke, N., Bewley, A., & Paulus, D. (2017). Simple online and realtime tracking with a deep association metric. In: ICIP.","DOI":"10.1109\/ICIP.2017.8296962"},{"key":"2887_CR2","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1155\/2008\/246309","volume":"2008","author":"K Bernardin","year":"2008","unstructured":"Bernardin, K., & Stiefelhagen, R. (2008). Evaluating multiple object tracking performance: the clear mot metrics. EURASIP Journal on Image and Video Processing, 2008, 1\u201310.","journal-title":"EURASIP Journal on Image and Video Processing"},{"key":"2887_CR3","doi-asserted-by":"crossref","unstructured":"Bras\u00f3, G., & Leal-Taix\u00e9, L. (2020). Learning a neural solver for multiple object tracking. CVPR.","DOI":"10.1109\/CVPR42600.2020.00628"},{"key":"2887_CR4","doi-asserted-by":"crossref","unstructured":"Cao, J., Weng, X., Khirodkar, R., Pang, J., & Kitani, K. (2023). Observation-centric sort: Rethinking sort for robust multi-object tracking. CVPR.","DOI":"10.1109\/CVPR52729.2023.00934"},{"key":"2887_CR5","unstructured":"Chaabane, M., Zhang, P., Beveridge, J.R., & O\u2019Hara, S. (2021). Deft: Detection embeddings for tracking. In: CVPRW"},{"key":"2887_CR6","doi-asserted-by":"crossref","unstructured":"Davison, A. J. (2003). Real-time simultaneous localisation and mapping with a single camera (p. ICCV)","DOI":"10.1109\/ICCV.2003.1238654"},{"key":"2887_CR7","doi-asserted-by":"crossref","unstructured":"Dendorfer, P., Yugay, V., Osep, A., & Leal-Taix\u00e9, L. (2022). Quo vadis: Is trajectory forecasting the key towards long-term multi-object tracking? In: NeurIPS","DOI":"10.52202\/068431-1139"},{"key":"2887_CR8","unstructured":"Eigen, D., Puhrsch, C., & Fergus, R. (2014). Depth map prediction from a single image using a multi-scale deep network. NeurIPS,"},{"key":"2887_CR9","doi-asserted-by":"crossref","unstructured":"Fan, L., Wang, F., Wang, N., & Zhang, Z. (2022). Fully sparse 3d object detection. In: NeurIPS.","DOI":"10.52202\/068431-0026"},{"key":"2887_CR10","doi-asserted-by":"crossref","unstructured":"Fischer, T., Pang, J., Huang, T. E., Qiu, L., Chen, H., Darrell, T., & Yu, F. (2022). Qdtrack: Quasi-dense similarity learning for appearance-only multiple object tracking arXiv:2210.06984.","DOI":"10.1109\/TPAMI.2023.3301975"},{"issue":"6","key":"2887_CR11","doi-asserted-by":"publisher","first-page":"381","DOI":"10.1145\/358669.358692","volume":"24","author":"MA Fischler","year":"1981","unstructured":"Fischler, M. A., & Bolles, R. C. (1981). Random sample consensus: a paradigm for model fitting with applications to image analysis and automated cartography. Communications of the ACM, 24(6), 381\u2013395.","journal-title":"Communications of the ACM"},{"key":"2887_CR12","doi-asserted-by":"crossref","unstructured":"Geiger, A., Lenz, P., & Urtasun, R. (2012). Are we ready for autonomous driving? the kitti vision benchmark suite. CVPR.","DOI":"10.1109\/CVPR.2012.6248074"},{"key":"2887_CR13","doi-asserted-by":"crossref","unstructured":"He, J., Chen, Y., Wang, N., & Zhang, Z. (2023). 3d video object detection with learnable object-centric global optimization. CVPR.","DOI":"10.1109\/CVPR52729.2023.00494"},{"key":"2887_CR14","doi-asserted-by":"crossref","unstructured":"He, J., Huang, Z., Wang, N., & Zhang, Z. (2021). Learnable graph matching: Incorporating graph partitioning with deep feature learning for multiple object tracking. CVPR.","DOI":"10.1109\/CVPR46437.2021.00526"},{"key":"2887_CR15","unstructured":"Hermans, A., Beyer, L., & Leibe, B. (2017). defense of the triplet loss for person re-identification arXiv:1703.07737."},{"key":"2887_CR16","unstructured":"Huang, J., Huang, G., Zhu, Z., & Du, D. (2021). Bevdet: High-performance multi-camera 3d object detection in bird-eye-view. arXiv:2112.11790."},{"key":"2887_CR17","doi-asserted-by":"crossref","unstructured":"Hu, H.-N., Cai, Q.-Z., Wang, D., Lin, J., Sun, M., Krahenbuhl, P., Darrell, T., & Yu, F. (2019). Joint monocular 3d vehicle detection and tracking (p. ICCV)","DOI":"10.1109\/ICCV.2019.00549"},{"issue":"2","key":"2887_CR18","doi-asserted-by":"publisher","first-page":"1992","DOI":"10.1109\/TPAMI.2022.3168781","volume":"45","author":"H-N Hu","year":"2022","unstructured":"Hu, H.-N., Yang, Y.-H., Fischer, T., Darrell, T., Yu, F., & Sun, M. (2022). Monocular quasi-dense 3d object tracking. IEEE Transactions on Pattern Analysis and Machine Intelligence, 45(2), 1992\u20132008.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2887_CR19","doi-asserted-by":"crossref","unstructured":"Kalman, R. E. (1960). A new approach to linear filtering and prediction problems","DOI":"10.1115\/1.3662552"},{"key":"2887_CR20","unstructured":"Kendall, A., & Gal, Y. (2017). What uncertainties do we need in bayesian deep learning for computer vision? In: NeurIPS"},{"key":"2887_CR21","doi-asserted-by":"crossref","unstructured":"Khurana, T., Dave, A., & Ramanan, D. (2021). Detecting invisible people (p. ICCV).","DOI":"10.1109\/ICCV48922.2021.00316"},{"key":"2887_CR22","unstructured":"Kingma, D. P., & Ba, J. (2014). Adam: A method for stochastic optimization arXiv:1412.6980."},{"key":"2887_CR23","doi-asserted-by":"crossref","unstructured":"Li, Y., Bao, H., Ge, Z., Yang, J., Sun, J., & Li, Z. (2022). Bevstereo: Enhancing depth estimation in multi-view 3d object detection with dynamic temporal stereo. arXiv:2209.10248","DOI":"10.1609\/aaai.v37i2.25234"},{"key":"2887_CR24","doi-asserted-by":"crossref","unstructured":"Li, Z., Wang, W., Li, H., Xie, E., Sima, C., Lu, T., Qiao, Y., & Dai, J. (2022). Bevformer: Learning bird\u2019s-eye-view representation from multi-camera images via spatiotemporal transformers. In: ECCV.","DOI":"10.1007\/978-3-031-20077-9_1"},{"key":"2887_CR25","doi-asserted-by":"publisher","first-page":"548","DOI":"10.1007\/s11263-020-01375-2","volume":"129","author":"J Luiten","year":"2021","unstructured":"Luiten, J., Osep, A., Dendorfer, P., Torr, P., Geiger, A., Leal-Taix\u00e9, L., & Leibe, B. (2021). Hota: A higher order metric for evaluating multi-object tracking. International journal of computer vision, 129, 548\u2013578.","journal-title":"International journal of computer vision"},{"issue":"4","key":"2887_CR26","doi-asserted-by":"publisher","first-page":"71","DOI":"10.1145\/3386569.3392377","volume":"39","author":"X Luo","year":"2020","unstructured":"Luo, X., Huang, J.-B., Szeliski, R., Matzen, K., & Kopf, J. (2020). Consistent video depth estimation. ACM Transactions on Graphics (ToG), 39(4), 71\u20131.","journal-title":"ACM Transactions on Graphics (ToG)"},{"key":"2887_CR27","doi-asserted-by":"crossref","unstructured":"Lu, Z., Rathod, V., Votel, R., & Huang, J. (2020). Retinatrack: Online single stage joint detection and tracking. CVPR.","DOI":"10.1109\/CVPR42600.2020.01468"},{"key":"2887_CR28","doi-asserted-by":"crossref","unstructured":"Meinhardt, T., Kirillov, A., Leal-Taixe, L., & Feichtenhofer, C. (2022). Trackformer: Multi-object tracking with transformers. CVPR.","DOI":"10.1109\/CVPR52688.2022.00864"},{"issue":"5","key":"2887_CR29","doi-asserted-by":"publisher","first-page":"1147","DOI":"10.1109\/TRO.2015.2463671","volume":"31","author":"R Mur-Artal","year":"2015","unstructured":"Mur-Artal, R., Montiel, J. M. M., & Tardos, J. D. (2015). Orb-slam: a versatile and accurate monocular slam system. IEEE transactions on robotics, 31(5), 1147\u20131163.","journal-title":"IEEE transactions on robotics"},{"key":"2887_CR30","doi-asserted-by":"crossref","unstructured":"Newcombe, R. A., Lovegrove, S. J., & Davison, A. J. (2011). Dtam: Dense tracking and mapping in real-time (p. ICCV)","DOI":"10.1109\/ICCV.2011.6126513"},{"key":"2887_CR31","doi-asserted-by":"crossref","unstructured":"Pang, Z., Li, J., Tokmakov, P., Chen, D., Zagoruyko, S., & Wang, Y.-X. (2023). Standing between past and future: Spatio-temporal modeling for multi-camera 3d multi-object tracking. CVPR.","DOI":"10.1109\/CVPR52729.2023.01719"},{"key":"2887_CR32","doi-asserted-by":"crossref","unstructured":"Park, D., Ambrus, R., Guizilini, V., Li, J., & Gaidon, A. (2021). Is pseudo-lidar needed for monocular 3d object detection? In: ICCV","DOI":"10.1109\/ICCV48922.2021.00313"},{"key":"2887_CR33","unstructured":"Paszke, A., Gross, S., Massa, F., Lerer, A., Bradbury, J., Chanan, G., Killeen, T., Lin, Z., Gimelshein, N., & Antiga, L. (2019). Pytorch: An imperative style, high-performance deep learning library. NeurIPS."},{"key":"2887_CR34","doi-asserted-by":"crossref","unstructured":"Ranftl, R., Lasinger, K., Hafner, D., Schindler, K., & Koltun, V. (2022). Towards robust monocular depth estimation: Mixing datasets for zero-shot cross-dataset transfer. IEEE Transactions on Pattern Analysis and Machine Intelligence, 44(3),","DOI":"10.1109\/TPAMI.2020.3019967"},{"key":"2887_CR35","unstructured":"Rangesh, A., Maheshwari, P., Gebre, M., Mhatre, S., Ramezani, V., & Trivedi, M. M. (2021). Trackmpnn: A message passing graph neural architecture for multi-object tracking arXiv:2101.04206."},{"key":"2887_CR36","doi-asserted-by":"crossref","unstructured":"Ristani, E., Solera, F., Zou, R., Cucchiara, R., & Tomasi, C. (2016). Performance measures and a data set for multi-target, multi-camera tracking. ECCV.","DOI":"10.1007\/978-3-319-48881-3_2"},{"key":"2887_CR37","doi-asserted-by":"crossref","unstructured":"Saleh, F., Aliakbarian, S., Rezatofighi, H., Salzmann, M., & Gould, S. (2021). Probabilistic tracklet scoring and inpainting for multiple object tracking. CVPR.","DOI":"10.1109\/CVPR46437.2021.01410"},{"key":"2887_CR38","doi-asserted-by":"crossref","unstructured":"Sch\u00f6nberger, J. L., & Frahm, J.-M. (2016). Structure-from-motion revisited. CVPR.","DOI":"10.1109\/CVPR.2016.445"},{"key":"2887_CR39","doi-asserted-by":"crossref","unstructured":"Sch\u00f6nberger, J. L., Zheng, E., Pollefeys, M., & Frahm, J.-M. (2016). Pixelwise view selection for unstructured multi-view stereo. ECCV.","DOI":"10.1007\/978-3-319-46487-9_31"},{"key":"2887_CR40","doi-asserted-by":"crossref","unstructured":"Shi, S., Guo, C., Jiang, L., Wang, Z., Shi, J., Wang, X., & Li, H. (2020). Pv-rcnn: Point-voxel feature set abstraction for 3d object detection. In: CVPR","DOI":"10.1109\/CVPR42600.2020.01054"},{"key":"2887_CR41","doi-asserted-by":"crossref","unstructured":"Tian, Z., Shen, C., Chen, H., & He, T. (2019). Fcos: Fully convolutional one-stage object detection (p. ICCV)","DOI":"10.1109\/ICCV.2019.00972"},{"key":"2887_CR42","unstructured":"Tokmakov, P., Jabri, A., Li, J., & Gaidon, A. (2022). Object permanence emerges in a random walk along memory (p. ICML)"},{"key":"2887_CR43","doi-asserted-by":"crossref","unstructured":"Tokmakov, P., Li, J., Burgard, W., & Gaidon, A. (2021). Learning to track with object permanence (p. ICCV)","DOI":"10.1109\/ICCV48922.2021.01068"},{"key":"2887_CR44","doi-asserted-by":"crossref","unstructured":"Wang, Q., Chen, Y., Pang, Z., Wang, N., & Zhang, Z. (2021). Immortal tracker: Tracklet never dies. arXiv:2111.13672.","DOI":"10.31219\/osf.io\/nw3fy"},{"key":"2887_CR45","unstructured":"Wang, Y., Guizilini, V.C., Zhang, T., Wang, Y., Zhao, H., & Solomon, J. (2022). Detr3d: 3d object detection from multi-view images via 3d-to-2d queries. In: CoRL."},{"key":"2887_CR46","doi-asserted-by":"crossref","unstructured":"Wang, G., Gu, R., Liu, Z., Hu, W., Song, M., & Hwang, J.-N. (2021). Track without appearance: Learn box and tracklet embedding with local and global motion patterns for vehicle tracking (p. ICCV)","DOI":"10.1109\/ICCV48922.2021.00973"},{"key":"2887_CR47","doi-asserted-by":"crossref","unstructured":"Wang, T., Pang, J., & Lin, D. (2022). Monocular 3d object detection with depth from motion. ECCV.","DOI":"10.1007\/978-3-031-20077-9_23"},{"key":"2887_CR48","doi-asserted-by":"crossref","unstructured":"Wang, Q., Ye, V., Gao, H., Zeng, W., Austin, J., Li, Z., & Kanazawa, A. (2025). Shape of motion: 4d reconstruction from a single video. Proceedings of the IEEE\/CVF International Conference on Computer Vision (pp. 9660\u20139672)","DOI":"10.1109\/ICCV51701.2025.00901"},{"key":"2887_CR49","doi-asserted-by":"crossref","unstructured":"Wang, T., Zhu, X., Pang, J., & Lin, D. (2021). Fcos3d: Fully convolutional one-stage monocular 3d object detection (p. ICCV)","DOI":"10.1109\/ICCVW54120.2021.00107"},{"key":"2887_CR50","doi-asserted-by":"crossref","unstructured":"Weng, X., Wang, Y., Man, Y., & Kitani, K.M. (2020). Gnn3dmot: Graph neural network for 3d multi-object tracking with 2d-3d multi-feature learning. In: CVPR.","DOI":"10.1109\/CVPR42600.2020.00653"},{"key":"2887_CR51","doi-asserted-by":"crossref","unstructured":"Weng, X., Wang, J., Held, D., & Kitani, K. (2020). Ab3dmot: A baseline for 3d multi-object tracking and new evaluation metrics arXiv:2008.08063.","DOI":"10.1109\/IROS45743.2020.9341164"},{"key":"2887_CR52","doi-asserted-by":"crossref","unstructured":"Wojke, N., Bewley, A., & Paulus, D. (2017). Simple online and realtime tracking with a deep association metric. In: ICIP.","DOI":"10.1109\/ICIP.2017.8296962"},{"key":"2887_CR53","doi-asserted-by":"crossref","unstructured":"Xu, Y., Osep, A., Ban, Y., Horaud, R., Leal-Taix\u00e9, L., & Alameda-Pineda, X. (2020). How to train your deep multi-object tracker. CVPR.","DOI":"10.1109\/CVPR42600.2020.00682"},{"key":"2887_CR54","doi-asserted-by":"crossref","unstructured":"Yu, F., Wang, D., Shelhamer, E., & Darrell, T. (2018). Deep layer aggregation. CVPR.","DOI":"10.1109\/CVPR.2018.00255"},{"key":"2887_CR55","doi-asserted-by":"crossref","unstructured":"Zeng, F., Dong, B., Zhang, Y., Wang, T., Zhang, X., & Wei, Y. (2022). Motr: End-to-end multiple-object tracking with transformer. In: ECCV.","DOI":"10.1007\/978-3-031-19812-0_38"},{"key":"2887_CR56","doi-asserted-by":"crossref","unstructured":"Zhang, T., Chen, X., Wang, Y., Wang, Y., & Zhao, H. (2022). Mutr3d: A multi-camera tracking framework via 3d-to-2d queries. In: CVPR.","DOI":"10.1109\/CVPRW56347.2022.00500"},{"key":"2887_CR57","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Sun, P., Jiang, Y., Yu, D., Weng, F., Yuan, Z., Luo, P., Liu, W., & Wang, X. (2022). Bytetrack: Multi-object tracking by associating every detection box. In: ECCV.","DOI":"10.1007\/978-3-031-20047-2_1"},{"issue":"4","key":"2887_CR58","first-page":"1","volume":"40","author":"Z Zhang","year":"2021","unstructured":"Zhang, Z., Cole, F., Tucker, R., Freeman, W. T., & Dekel, T. (2021). Consistent depth of moving objects in video. ACM Transactions on Graphics (TOG), 40(4), 1\u201312.","journal-title":"ACM Transactions on Graphics (TOG)"},{"key":"2887_CR59","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Lu, J., & Zhou, J. (2021). Objects are different: Flexible monocular 3d object detection. CVPR.","DOI":"10.1109\/CVPR46437.2021.00330"},{"key":"2887_CR60","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Wang, C., Wang, X., Zeng, W., & Liu, W. (2021). Fairmot: On the fairness of detection and re-identification in multiple object tracking. International Journal of Computer Vision,129(11), 3069\u20133087.","DOI":"10.1007\/s11263-021-01513-4"},{"key":"2887_CR61","doi-asserted-by":"crossref","unstructured":"Zhou, T., Brown, M., Snavely, N., & Lowe, D.G. (2017). Unsupervised learning of depth and ego-motion from video. In: CVPR","DOI":"10.1109\/CVPR.2017.700"},{"key":"2887_CR62","doi-asserted-by":"crossref","unstructured":"Zhou, X., Koltun, V., & Kr\u00e4henb\u00fchl, P. (2020). Tracking objects as points. ECCV.","DOI":"10.1007\/978-3-030-58548-8_28"},{"key":"2887_CR63","unstructured":"Zhou, X., Wang, D., & Kr\u00e4henb\u00fchl, P. (2019). Objects as points arXiv:1904.07850."},{"key":"2887_CR64","doi-asserted-by":"crossref","unstructured":"Zhou, X., Yin, T., Koltun, V., & Kr\u00e4henb\u00fchl, P. (2022). Global tracking transformers. CVPR.","DOI":"10.1109\/CVPR52688.2022.00857"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-026-02887-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-026-02887-z","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-026-02887-z.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,29]],"date-time":"2026-07-29T05:50:21Z","timestamp":1785304221000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-026-02887-z"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6,20]]},"references-count":64,"journal-issue":{"issue":"7","published-print":{"date-parts":[[2026,7]]}},"alternative-id":["2887"],"URL":"https:\/\/doi.org\/10.1007\/s11263-026-02887-z","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,6,20]]},"assertion":[{"value":"3 April 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 May 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"20 June 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"325"}}