{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T04:08:38Z","timestamp":1750219718734,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":39,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,5,19]],"date-time":"2023-05-19T00:00:00Z","timestamp":1684454400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62006211"],"award-info":[{"award-number":["62006211"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"China Postdoctoral Science Foundation","award":["2019TQ0286,2020M682349"],"award-info":[{"award-number":["2019TQ0286,2020M682349"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,5,19]]},"DOI":"10.1145\/3604078.3604079","type":"proceedings-article","created":{"date-parts":[[2023,10,26]],"date-time":"2023-10-26T16:46:07Z","timestamp":1698338767000},"page":"1-9","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Action Recognition Based on Dense Action Captioning"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-3135-0591","authenticated-orcid":false,"given":"Chenglong","family":"Zhang","sequence":"first","affiliation":[{"name":"School of Computer and Artificial Intelligence \/ Zhengzhou University \/ Natural language processing Laboratory, Zhengzhou University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6418-6284","authenticated-orcid":false,"given":"Lijuan","family":"Zhou","sequence":"additional","affiliation":[{"name":"School of Computer and Artificial Intelligence \/ Zhengzhou University \/ Natural language processing Laboratory, Zhengzhou University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-1970-3437","authenticated-orcid":false,"given":"Changyong","family":"Niu","sequence":"additional","affiliation":[{"name":"School of Computer and Artificial Intelligence \/ Zhengzhou University, Zhengzhou University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,10,26]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/1922649.1922653"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2015.03.006"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.5244\/C.29.177"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.619"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-04114-8_40"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2015.11.019"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.14257\/ijsip.2014.7.3.10"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2019.2890829"},{"key":"e_1_3_2_1_9_1","volume-title":"An Iterative Classification Network for Semantic Action Recognition. In 2021 The 5th International Conference on Video and Image Processing. 32\u201337","author":"Zhou Lijuan","year":"2021","unstructured":"Lijuan Zhou , Junfu Chen , and Xiaojie Qian . 2021 . An Iterative Classification Network for Semantic Action Recognition. In 2021 The 5th International Conference on Video and Image Processing. 32\u201337 . Lijuan Zhou, Junfu Chen, and Xiaojie Qian. 2021. An Iterative Classification Network for Semantic Action Recognition. In 2021 The 5th International Conference on Video and Image Processing. 32\u201337."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475253"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patrec.2022.06.012"},{"key":"e_1_3_2_1_12_1","volume-title":"Tell me what you see: A zero-shot action recognition method based on natural language descriptions. arXiv preprint arXiv:2112.09976","author":"Estevam Valter","year":"2021","unstructured":"Valter Estevam , Rayson Laroca , David Menotti , and Helio Pedrini . 2021. Tell me what you see: A zero-shot action recognition method based on natural language descriptions. arXiv preprint arXiv:2112.09976 ( 2021 ). Valter Estevam, Rayson Laroca, David Menotti, and Helio Pedrini. 2021. Tell me what you see: A zero-shot action recognition method based on natural language descriptions. arXiv preprint arXiv:2112.09976 (2021)."},{"volume-title":"Research on Skeleton-based Action Recognition with Textual Semantics. Master's thesis","author":"Zhang Weicong","key":"e_1_3_2_1_13_1","unstructured":"Weicong Zhang . 2022. Research on Skeleton-based Action Recognition with Textual Semantics. Master's thesis . Zhengzhou University . Weicong Zhang. 2022. Research on Skeleton-based Action Recognition with Textual Semantics. Master's thesis. Zhengzhou University."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3512576.3512606"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2017.06.035"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00081"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00807"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00193"},{"key":"e_1_3_2_1_19_1","volume-title":"3D human action recognition with siamese-LSTM based deep metric learning. arXiv preprint arXiv:1807.02131","author":"Yucer Seyma","year":"2018","unstructured":"Seyma Yucer and Yusuf Sinan Akgul . 2018. 3D human action recognition with siamese-LSTM based deep metric learning. arXiv preprint arXiv:1807.02131 ( 2018 ). Seyma Yucer and Yusuf Sinan Akgul. 2018. 3D human action recognition with siamese-LSTM based deep metric learning. arXiv preprint arXiv:1807.02131 (2018)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.18178\/joig.6.2.174-180"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.12720\/joig.2.1.28-32"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.18178\/joig.3.2.96-101"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1007\/s00138-022-01328-4"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1007\/s00138-021-01249-8"},{"key":"e_1_3_2_1_25_1","volume-title":"A comprehensive study of deep video action recognition. arXiv preprint arXiv:2012.06567","author":"Zhu Yi","year":"2020","unstructured":"Yi Zhu , Xinyu Li , Chunhui Liu , Mohammadreza Zolfaghari , Yuanjun Xiong , Chongruo Wu , Zhi Zhang , Joseph Tighe , R Manmatha , and Mu Li. 2020. A comprehensive study of deep video action recognition. arXiv preprint arXiv:2012.06567 ( 2020 ). Yi Zhu, Xinyu Li, Chunhui Liu, Mohammadreza Zolfaghari, Yuanjun Xiong, Chongruo Wu, Zhi Zhang, Joseph Tighe, R Manmatha, and Mu Li. 2020. A comprehensive study of deep video action recognition. arXiv preprint arXiv:2012.06567 (2020)."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.imavis.2017.01.010"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.3390\/s19051005"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2011.6126386"},{"key":"e_1_3_2_1_29_1","volume-title":"Actionclip: A new paradigm for video action recognition. arXiv preprint arXiv:2109.08472","author":"Wang Mengmeng","year":"2021","unstructured":"Mengmeng Wang , Jiazheng Xing , and Yong Liu . 2021 . Actionclip: A new paradigm for video action recognition. arXiv preprint arXiv:2109.08472 (2021). Mengmeng Wang, Jiazheng Xing, and Yong Liu. 2021. Actionclip: A new paradigm for video action recognition. arXiv preprint arXiv:2109.08472 (2021)."},{"key":"e_1_3_2_1_30_1","volume-title":"Translating videos to natural language using deep recurrent neural networks. arXiv preprint arXiv:1412.4729","author":"Venugopalan Subhashini","year":"2014","unstructured":"Subhashini Venugopalan , Huijuan Xu , Jeff Donahue , Marcus Rohrbach , Raymond Mooney , and Kate Saenko . 2014. Translating videos to natural language using deep recurrent neural networks. arXiv preprint arXiv:1412.4729 ( 2014 ). Subhashini Venugopalan, Huijuan Xu, Jeff Donahue, Marcus Rohrbach, Raymond Mooney, and Kate Saenko. 2014. Translating videos to natural language using deep recurrent neural networks. arXiv preprint arXiv:1412.4729 (2014)."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00854"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v35i3.16353"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.83"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2020.3014606"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00677"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW54120.2021.00356"},{"key":"e_1_3_2_1_37_1","volume-title":"Proceedings of the 40th annual meeting of the Association for Computational Linguistics. 311\u2013318","author":"Papineni Kishore","year":"2002","unstructured":"Kishore Papineni , Salim Roukos , Todd Ward , and Wei-Jing Zhu . 2002 . Bleu: a method for automatic evaluation of machine translation . In Proceedings of the 40th annual meeting of the Association for Computational Linguistics. 311\u2013318 . Kishore Papineni, Salim Roukos, Todd Ward, and Wei-Jing Zhu. 2002. Bleu: a method for automatic evaluation of machine translation. In Proceedings of the 40th annual meeting of the Association for Computational Linguistics. 311\u2013318."},{"key":"e_1_3_2_1_38_1","volume-title":"Proceedings of the acl workshop on intrinsic and extrinsic evaluation measures for machine translation and\/or summarization. 65\u201372","author":"Banerjee Satanjeev","year":"2005","unstructured":"Satanjeev Banerjee and Alon Lavie . 2005 . METEOR: An automatic metric for MT evaluation with improved correlation with human judgments . In Proceedings of the acl workshop on intrinsic and extrinsic evaluation measures for machine translation and\/or summarization. 65\u201372 . Satanjeev Banerjee and Alon Lavie. 2005. METEOR: An automatic metric for MT evaluation with improved correlation with human judgments. In Proceedings of the acl workshop on intrinsic and extrinsic evaluation measures for machine translation and\/or summarization. 65\u201372."},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7299087"}],"event":{"name":"ICDIP 2023: The 15th International Conference on Digital Image Processing","acronym":"ICDIP 2023","location":"Nanjing China"},"container-title":["Proceedings of the 15th International Conference on Digital Image Processing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3604078.3604079","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3604078.3604079","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T16:36:16Z","timestamp":1750178176000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3604078.3604079"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,5,19]]},"references-count":39,"alternative-id":["10.1145\/3604078.3604079","10.1145\/3604078"],"URL":"https:\/\/doi.org\/10.1145\/3604078.3604079","relation":{},"subject":[],"published":{"date-parts":[[2023,5,19]]},"assertion":[{"value":"2023-10-26","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}