{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T15:15:31Z","timestamp":1784301331348,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":58,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/https:\/\/doi.org\/10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["No. 62376282"],"award-info":[{"award-number":["No. 62376282"]}],"id":[{"id":"10.13039\/https:\/\/doi.org\/10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3664647.3680944","type":"proceedings-article","created":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T06:59:41Z","timestamp":1729925981000},"page":"4082-4091","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":28,"title":["MambaTrack: A Simple Baseline for Multiple Object Tracking with State Space Model"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-4707-343X","authenticated-orcid":false,"given":"Changcheng","family":"Xiao","sequence":"first","affiliation":[{"name":"National University of Defense Technology, Changsha, Hunan Province, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8750-3505","authenticated-orcid":false,"given":"Qiong","family":"Cao","sequence":"additional","affiliation":[{"name":"JD Explore Academy, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7552-201X","authenticated-orcid":false,"given":"Zhigang","family":"Luo","sequence":"additional","affiliation":[{"name":"National University of Defense Technology, Changsha, Hunan Province, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4238-8985","authenticated-orcid":false,"given":"Long","family":"Lan","sequence":"additional","affiliation":[{"name":"National University of Defense Technology, Changsha, Hunan Province, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Jamie Ryan Kiros, and Geoffrey E Hinton","author":"Ba Jimmy Lei","year":"2016","unstructured":"Jimmy Lei Ba, Jamie Ryan Kiros, and Geoffrey E Hinton. 2016. Layer normalization. arXiv preprint arXiv:1607.06450 (2016)."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00103"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICIP.2016.7533003"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00628"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01164"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00934"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"e_1_3_2_1_8_1","volume-title":"DEFT: Detection Embeddings for Tracking. arXiv preprint arXiv:2102.02267","author":"Chaabane Mohamed","year":"2021","unstructured":"Mohamed Chaabane, Peter Zhang, Ross Beveridge, and Stephen O'Hara. 2021. DEFT: Detection Embeddings for Tracking. arXiv preprint arXiv:2102.02267 (2021)."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00910"},{"key":"e_1_3_2_1_10_1","volume-title":"Mot20: A benchmark for multi object tracking in crowded scenes. arXiv preprint arXiv:2003.09003","author":"Dendorfer Patrick","year":"2020","unstructured":"Patrick Dendorfer, Hamid Rezatofighi, Anton Milan, Javen Shi, Daniel Cremers, Ian Reid, Stefan Roth, Konrad Schindler, and Laura Leal-Taix\u00e9. 2020. Mot20: A benchmark for multi object tracking in crowded scenes. arXiv preprint arXiv:2003.09003 (2020)."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2020.3005662"},{"key":"e_1_3_2_1_12_1","volume-title":"Yolox: Exceeding yolo series in","author":"Ge Zheng","year":"2021","unstructured":"Zheng Ge, Songtao Liu, Feng Wang, Zeming Li, and Jian Sun. 2021. Yolox: Exceeding yolo series in 2021. arXiv preprint arXiv:2107.08430 (2021)."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2012.6248074"},{"key":"e_1_3_2_1_14_1","volume-title":"Mamba: Linear-time sequence modeling with selective state spaces. arXiv preprint arXiv:2312.00752","author":"Gu Albert","year":"2023","unstructured":"Albert Gu and Tri Dao. 2023. Mamba: Linear-time sequence modeling with selective state spaces. arXiv preprint arXiv:2312.00752 (2023)."},{"key":"e_1_3_2_1_15_1","volume-title":"Hippo: Recurrent memory with optimal polynomial projections.","author":"Gu Albert","year":"2020","unstructured":"Albert Gu, Tri Dao, Stefano Ermon, Atri Rudra, and Christopher R\u00e9. 2020. Hippo: Recurrent memory with optimal polynomial projections."},{"key":"e_1_3_2_1_16_1","unstructured":"Albert Gu Karan Goel and Christopher Re. 2021. Efficiently Modeling Long Sequences with Structured State Spaces."},{"key":"e_1_3_2_1_17_1","unstructured":"Albert Gu Isys Johnson Karan Goel Khaled Saab Tri Dao Atri Rudra and Christopher R\u00e9. 2021. Combining recurrent convolutional and continuous-time models with linear state space layers."},{"key":"e_1_3_2_1_18_1","volume-title":"Long short-term memory. Neural computation","author":"Hochreiter Sepp","year":"1997","unstructured":"Sepp Hochreiter and J\u00fcrgen Schmidhuber. 1997. Long short-term memory. Neural computation, Vol. 9, 8 (1997), 1735--1780."},{"key":"e_1_3_2_1_19_1","unstructured":"Hsiang-Wei Huang Cheng-Yen Yang Wenhao Chai Zhongyu Jiang and Jenq-Neng Hwang. 2024. Exploring Learning-based Motion Models in Multi-Object Tracking. arxiv: 2403.10826 [cs.CV] https:\/\/arxiv.org\/abs\/2403.10826"},{"key":"e_1_3_2_1_20_1","unstructured":"Md Mohaiminul Islam and Gedas Bertasius. 2022. Long movie clip classification with state-space video models."},{"key":"e_1_3_2_1_21_1","volume-title":"Tony Braskich, and Gedas Bertasius.","author":"Islam Md Mohaiminul","year":"2023","unstructured":"Md Mohaiminul Islam, Mahmudul Hasan, Kishan Shamsundar Athrey, Tony Braskich, and Gedas Bertasius. 2023. Efficient Movie Scene Detection using State-Space Transformers."},{"key":"e_1_3_2_1_22_1","unstructured":"Rudolf Emil Kalman et al. 1960. Contributions to the theory of optimal control. Bol. soc. mat. mexicana Vol. 5 2 (1960) 102--119."},{"key":"e_1_3_2_1_23_1","volume-title":"Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980","author":"Kingma Diederik P","year":"2014","unstructured":"Diederik P Kingma and Jimmy Ba. 2014. Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014)."},{"key":"e_1_3_2_1_24_1","volume-title":"The Hungarian method for the assignment problem. Naval research logistics quarterly","author":"Kuhn Harold W","year":"1955","unstructured":"Harold W Kuhn. 1955. The Hungarian method for the assignment problem. Naval research logistics quarterly, Vol. 2, 1--2 (1955), 83--97."},{"key":"e_1_3_2_1_25_1","unstructured":"Long Lan Dacheng Tao Chen Gong Naiyang Guan and Zhigang Luo. 2016. Online Multi-Object Tracking by Quadratic Pseudo-Boolean Optimization.. In IJCAI. 3396--3402."},{"key":"e_1_3_2_1_26_1","unstructured":"Kunchang Li Xinhao Li Yi Wang Yinan He Yali Wang Limin Wang and Yu Qiao. 2024. VideoMamba: State Space Model for Efficient Video Understanding. arxiv: 2403.06977 [cs.CV]"},{"key":"e_1_3_2_1_27_1","unstructured":"Yuhong Li Tianle Cai Yi Zhang Deming Chen and Debadeepta Dey. 2022. What Makes Convolutional Models Great on Long Sequence Modeling?"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"crossref","unstructured":"Chen Long Ai Haizhou Zhuang Zijie and Shang Chong. 2018. Real-time Multiple People Tracking with Deeply Learned Candidate Selection and Person Re-identification. In ICME.","DOI":"10.1109\/ICME.2018.8486597"},{"key":"e_1_3_2_1_29_1","volume-title":"Hota: A higher order metric for evaluating multi-object tracking. International journal of computer vision","author":"Luiten Jonathon","year":"2021","unstructured":"Jonathon Luiten, Aljosa Osep, Patrick Dendorfer, Philip Torr, Andreas Geiger, Laura Leal-Taix\u00e9, and Bastian Leibe. 2021. Hota: A higher order metric for evaluating multi-object tracking. International journal of computer vision, Vol. 129, 2 (2021), 548--578."},{"key":"e_1_3_2_1_30_1","volume-title":"Jrdb: A dataset and benchmark of egocentric robot visual perception of humans in built environments","author":"Martin-Martin Roberto","year":"2021","unstructured":"Roberto Martin-Martin, Mihir Patel, Hamid Rezatofighi, Abhijeet Shenoi, JunYoung Gwak, Eric Frankel, Amir Sadeghian, and Silvio Savarese. 2021. Jrdb: A dataset and benchmark of egocentric robot visual perception of humans in built environments. IEEE transactions on pattern analysis and machine intelligence (2021)."},{"key":"e_1_3_2_1_31_1","unstructured":"Harsh Mehta Ankit Gupta Ashok Cutkosky and Behnam Neyshabur. 2022. Long Range Language Modeling via Gated State Spaces."},{"key":"e_1_3_2_1_32_1","volume-title":"MOT16: A benchmark for multi-object tracking. arXiv preprint arXiv:1603.00831","author":"Milan Anton","year":"2016","unstructured":"Anton Milan, Laura Leal-Taix\u00e9, Ian Reid, Stefan Roth, and Konrad Schindler. 2016. MOT16: A benchmark for multi-object tracking. arXiv preprint arXiv:1603.00831 (2016)."},{"key":"e_1_3_2_1_33_1","unstructured":"Eric Nguyen Karan Goel Albert Gu Gordon Downs Preey Shah Tri Dao Stephen Baccus and Christopher R\u00e9. 2022. S4nd: Modeling images and videos as multidimensional signals with state spaces."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00023"},{"key":"e_1_3_2_1_35_1","volume-title":"Moe-mamba: Efficient selective state space models with mixture of experts. arXiv preprint arXiv:2401.04081","author":"Pi\u00f3ro Maciej","year":"2024","unstructured":"Maciej Pi\u00f3ro, Kamil Ciebiera, Krystian Kr\u00f3l, Jan Ludziejewski, and Sebastian Jaszczur. 2024. Moe-mamba: Efficient selective state space models with mixture of experts. arXiv preprint arXiv:2401.04081 (2024)."},{"key":"e_1_3_2_1_36_1","volume-title":"Yolov3: An incremental improvement. arXiv preprint arXiv:1804.02767","author":"Redmon Joseph","year":"2018","unstructured":"Joseph Redmon and Ali Farhadi. 2018. Yolov3: An incremental improvement. arXiv preprint arXiv:1804.02767 (2018)."},{"key":"e_1_3_2_1_37_1","volume-title":"Faster r-cnn: Towards real-time object detection with region proposal networks. Advances in neural information processing systems","author":"Ren Shaoqing","year":"2015","unstructured":"Shaoqing Ren, Kaiming He, Ross Girshick, and Jian Sun. 2015. Faster r-cnn: Towards real-time object detection with region proposal networks. Advances in neural information processing systems, Vol. 28 (2015)."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-48881-3_2"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-48881-3_2"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01410"},{"key":"e_1_3_2_1_41_1","unstructured":"Jimmy TH Smith Andrew Warrington and Scott Linderman. 2022. Simplified State Space Layers for Sequence Modeling."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.02032"},{"key":"e_1_3_2_1_43_1","volume-title":"Transtrack: Multiple object tracking with transformer. arXiv preprint arXiv:2012.15460","author":"Sun Peize","year":"2020","unstructured":"Peize Sun, Jinkun Cao, Yi Jiang, Rufeng Zhang, Enze Xie, Zehuan Yuan, Changhu Wang, and Ping Luo. 2020. Transtrack: Multiple object tracking with transformer. arXiv preprint arXiv:2012.15460 (2020)."},{"key":"e_1_3_2_1_44_1","volume-title":"Attention is all you need. Advances in neural information processing systems","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems, Vol. 30 (2017)."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00813"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-67835-7_5"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58621-8_7"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICIP.2017.8296962"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01217"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2024.106539"},{"key":"e_1_3_2_1_51_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00846"},{"key":"e_1_3_2_1_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01076"},{"key":"e_1_3_2_1_53_1","volume-title":"MOTR: End-to-End Multiple-Object Tracking with TRansformer. In European Conference on Computer Vision (ECCV).","author":"Zeng Fangao","year":"2022","unstructured":"Fangao Zeng, Bin Dong, Yuang Zhang, Tiancai Wang, Xiangyu Zhang, and Yichen Wei. 2022. MOTR: End-to-End Multiple-Object Tracking with TRansformer. In European Conference on Computer Vision (ECCV)."},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20047-2_1"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-021-01513-4"},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58548-8_28"},{"key":"e_1_3_2_1_57_1","unstructured":"Xingyi Zhou Dequan Wang and Philipp Kr\u00e4henb\u00fchl. 2019. Objects as Points. In arXiv preprint arXiv:1904.07850."},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00857"}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","location":"Melbourne VIC Australia","acronym":"MM '24","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 32nd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3680944","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3664647.3680944","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:17:34Z","timestamp":1750295854000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3680944"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":58,"alternative-id":["10.1145\/3664647.3680944","10.1145\/3664647"],"URL":"https:\/\/doi.org\/10.1145\/3664647.3680944","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}