{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T05:05:52Z","timestamp":1765343152676,"version":"3.46.0"},"publisher-location":"New York, NY, USA","reference-count":62,"publisher":"ACM","funder":[{"name":"Nanjing University-China Mobile Communications Group Co.,Ltd. Joint Institute"},{"name":"Nanjing Key S&T Special Projects","award":["202309006"],"award-info":[{"award-number":["202309006"]}]},{"name":"NSFC","award":["62202233"],"award-info":[{"award-number":["62202233"]}]},{"name":"Grant from State Key Laboratory for Novel Software Technology, Nanjing University","award":["KFKT2024B18"],"award-info":[{"award-number":["KFKT2024B18"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755186","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T07:37:21Z","timestamp":1761377841000},"page":"11805-11814","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Decode-What-Matters: Frame-Level Parallel Generative Decoding to Accelerate Large-Scale Video Analytics"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0003-7114-1417","authenticated-orcid":false,"given":"Xiaokun","family":"Wang","sequence":"first","affiliation":[{"name":"Nanjing University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7058-1272","authenticated-orcid":false,"given":"Yuting","family":"Yan","sequence":"additional","affiliation":[{"name":"Nanjing University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6581-6399","authenticated-orcid":false,"given":"Sheng","family":"Zhang","sequence":"additional","affiliation":[{"name":"Nanjing University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-8233-329X","authenticated-orcid":false,"given":"Andong","family":"Zhu","sequence":"additional","affiliation":[{"name":"Nanjing University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0722-1757","authenticated-orcid":false,"given":"Ning","family":"Chen","sequence":"additional","affiliation":[{"name":"Soochow University, Suzhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8680-1922","authenticated-orcid":false,"given":"Yu","family":"Chen","sequence":"additional","affiliation":[{"name":"Nanjing University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1625-7575","authenticated-orcid":false,"given":"Zhuzhong","family":"Qian","sequence":"additional","affiliation":[{"name":"Nanjing University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1467-4519","authenticated-orcid":false,"given":"Sanglu","family":"Lu","sequence":"additional","affiliation":[{"name":"Nanjing University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9251-4337","authenticated-orcid":false,"given":"Yu","family":"Liang","sequence":"additional","affiliation":[{"name":"Nanjing Normal University, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Proceedings of USENIX Symposium on Networked Systems Design and Implementation (NSDI). 933-951","author":"Agarwal Neil","year":"2023","unstructured":"Neil Agarwal and Ravi Netravali. 2023. Boggart: Towards General-Purpose acceleration of retrospective video analytics. In Proceedings of USENIX Symposium on Networked Systems Design and Implementation (NSDI). 933-951."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDE.2019.00132"},{"key":"e_1_3_2_1_3_1","unstructured":"AV1 Video Codec. 2025. https:\/\/aomedia.org\/specifications\/av1\/."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11042-014-2345-z"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11554-018-0840-6"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.engappai.2022.105095"},{"key":"e_1_3_2_1_7_1","volume-title":"Proceedings of the Second Conference on Machine Learning and Systems (MLSys). 406-417","author":"Canel Christopher","year":"2019","unstructured":"Christopher Canel, Thomas Kim, Giulio Zhou, Conglong Li, Hyeontaek Lim, David G Andersen, Michael Kaminsky, and Subramanya Dulloor. 2019. Scaling video analytics on constrained edge nodes. In Proceedings of the Second Conference on Machine Learning and Systems (MLSys). 406-417."},{"key":"e_1_3_2_1_8_1","volume-title":"Proceedings of USENIX Symposium on Networked Systems Design and Implementation (NSDI). 533-548","author":"Chen Bo","year":"2024","unstructured":"Bo Chen, Zhisheng Yan, Yinjie Zhang, Zhe Yang, and Klara Nahrstedt. 2024. LiFteR: Unleash Learned Codecs in Video Streaming with Loose Frame Referencing. In Proceedings of USENIX Symposium on Networked Systems Design and Implementation (NSDI). 533-548."},{"key":"e_1_3_2_1_9_1","volume-title":"Proceedings of ACM International Conference on Embedded Networked Sensor Systems (SenSys). 155-168","author":"Yu-Han Chen Tiffany","year":"2015","unstructured":"Tiffany Yu-Han Chen, Lenin Ravindranath, Shuo Deng, Paramvir Bahl, and Hari Balakrishnan. 2015. Glimpse: Continuous, real-time object recognition on mobile devices. In Proceedings of ACM International Conference on Embedded Networked Sensor Systems (SenSys). 155-168."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00242"},{"key":"e_1_3_2_1_11_1","unstructured":"FFmpeg. 2025. https:\/\/ffmpeg.org\/."},{"key":"e_1_3_2_1_12_1","unstructured":"Google's libvpx Official Github Repository. 2025. https:\/\/github.com\/webmproject\/ libvpx\/."},{"key":"e_1_3_2_1_13_1","unstructured":"H.265 Specification. 2025. https:\/\/www.itu.int\/rec\/T-REC-H.265."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM42981.2021.9488741"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.155"},{"key":"e_1_3_2_1_17_1","volume-title":"Proceedings of USENIX Annual Technical Conference (ATC). 707-722","author":"Hwang Jinwoo","year":"2022","unstructured":"Jinwoo Hwang, Minsu Kim, Daeun Kim, Seungho Nam, Yoonsung Kim, Dohee Kim, Hardik Sharma, and Jongse Park. 2022. CoVA: Exploiting Compressed-Domain analysis to accelerate video analytics. In Proceedings of USENIX Annual Technical Conference (ATC). 707-722."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.632"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/3230543.3230574"},{"key":"e_1_3_2_1_20_1","volume-title":"Blazeit: Optimizing declarative aggregation and limit queries for neural network-based video analytics. arXiv preprint arXiv:1805.01046","author":"Kang Daniel","year":"2018","unstructured":"Daniel Kang, Peter Bailis, and Matei Zaharia. 2018. Blazeit: Optimizing declarative aggregation and limit queries for neural network-based video analytics. arXiv preprint arXiv:1805.01046 (2018)."},{"key":"e_1_3_2_1_21_1","volume-title":"Noscope: optimizing neural network queries over video at scale. arXiv preprint arXiv:1703.02529","author":"Kang Daniel","year":"2017","unstructured":"Daniel Kang, John Emmons, Firas Abuzaid, Peter Bailis, and Matei Zaharia. 2017. Noscope: optimizing neural network queries over video at scale. arXiv preprint arXiv:1703.02529 (2017)."},{"key":"e_1_3_2_1_22_1","unstructured":"Kiloview. 2025. Kiloview D350\/D260 Multi-Channel Decoder. https:\/\/www.kiloview.com\/en\/decoder\/."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612585"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/3387514.3405874"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02236"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/3300061.3300116"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00153"},{"key":"e_1_3_2_1_28_1","volume-title":"Rtmdet: An empirical study of designing real-time object detectors. arXiv preprint arXiv:2212.07784","author":"Lyu Chengqi","year":"2022","unstructured":"Chengqi Lyu, Wenwei Zhang, Haian Huang, Yue Zhou, Yudong Wang, Yanyi Liu, Shilong Zhang, and Kai Chen. 2022. Rtmdet: An empirical study of designing real-time object detectors. arXiv preprint arXiv:2212.07784 (2022)."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCAS.2017.8050296"},{"key":"e_1_3_2_1_30_1","unstructured":"NVIDIA. 2025. NVIDIA Video Codec SDK. https:\/\/developer.nvidia.com\/video-codec-sdk."},{"key":"e_1_3_2_1_31_1","unstructured":"OpenMMLab. 2025. MMSegmentation. https:\/\/github.com\/open-mmlab\/mmsegmentation."},{"key":"e_1_3_2_1_32_1","unstructured":"OpenVINO. 2025. https:\/\/www.intel.com\/content\/www\/us\/en\/developer\/tools\/openvino-toolkit\/overview.html."},{"key":"e_1_3_2_1_33_1","unstructured":"Paddle. 2025. PaddleDetection. https:\/\/github.com\/PaddlePaddle."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/2872362.2872376"},{"key":"e_1_3_2_1_35_1","unstructured":"Pytorch. 2025. https:\/\/pytorch.org\/."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.3390\/info12010014"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICMA.2014.6885761"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00271"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.236"},{"key":"e_1_3_2_1_40_1","volume-title":"Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556","author":"Simonyan Karen","year":"2014","unstructured":"Karen Simonyan and Andrew Zisserman. 2014. Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556 (2014)."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/INFOCOM52122.2024.10621392"},{"key":"e_1_3_2_1_42_1","unstructured":"TensorRT. 2025. https:\/\/developer.nvidia.com\/tensorrt."},{"key":"e_1_3_2_1_43_1","unstructured":"TVM. 2025. https:\/\/tvm.apache.org\/."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00721"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCAD.2021.3120076"},{"key":"e_1_3_2_1_46_1","unstructured":"Webm Official Website. 2025. https:\/\/www.webmproject.org\/."},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00140"},{"key":"e_1_3_2_1_48_1","volume-title":"Edgeduet: Tiling small object detection for edge assisted autonomous mobile vision","author":"Yang Zheng","year":"2022","unstructured":"Zheng Yang, Xu Wang, Jiahang Wu, Yi Zhao, Qiang Ma, Xin Miao, Li Zhang, and Zimu Zhou. 2022. Edgeduet: Tiling small object detection for edge assisted autonomous mobile vision. IEEE\/ACM Transactions on Networking (TON) (2022)."},{"key":"e_1_3_2_1_49_1","volume-title":"Proceedings of USENIX Symposium on Operating Systems Design and Implementation (OSDI). 645-661","author":"Yeo Hyunho","year":"2018","unstructured":"Hyunho Yeo, Youngmok Jung, Jaehong Kim, Jinwoo Shin, and Dongsu Han. 2018. Neural adaptive content-aware internet video delivery. In Proceedings of USENIX Symposium on Operating Systems Design and Implementation (OSDI). 645-661."},{"key":"e_1_3_2_1_50_1","unstructured":"YouTube. 2025a. 4 Corners Camera Downtown. https:\/\/www.youtube.com\/watch?v=ByED80IKdIU."},{"key":"e_1_3_2_1_51_1","unstructured":"YouTube. 2025b. 4K Road traffic video for object detection and tracking. https:\/\/www.youtube.com\/watch?v=QuUxHIVUoaY."},{"key":"e_1_3_2_1_52_1","unstructured":"YouTube. 2025c. Ammanford Cam Wales. https:\/\/www.youtube.com\/watch?v=Uczhfukba_k."},{"key":"e_1_3_2_1_53_1","unstructured":"YouTube. 2025d. Grand Avenue Bridge in Glenwood Springs Live Camera. https:\/\/www.youtube.com\/watch?v=B0YjuKbVZ5w."},{"key":"e_1_3_2_1_54_1","unstructured":"YouTube. 2025 e. People Walking Stock Footage. https:\/\/www.youtube.com\/watch?v=bwJ-TNu0hGM."},{"key":"e_1_3_2_1_55_1","unstructured":"YouTube. 2025 f. Village of Tilton. https:\/\/www.youtube.com\/watch?v=5_XSYlAfJZM."},{"key":"e_1_3_2_1_56_1","doi-asserted-by":"publisher","DOI":"10.1145\/3495243.3517016"},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1145\/3603269.3604825"},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.measurement.2022.112371"},{"key":"e_1_3_2_1_59_1","doi-asserted-by":"publisher","DOI":"10.1109\/TC.2020.2970413"},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.1145\/3447993.3448628"},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2019.07.011"},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00887"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Dublin Ireland","acronym":"MM '25"},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755186","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T05:02:28Z","timestamp":1765342948000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755186"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":62,"alternative-id":["10.1145\/3746027.3755186","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755186","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}