{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,16]],"date-time":"2026-06-16T04:50:22Z","timestamp":1781585422499,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":79,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3754506","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T05:56:43Z","timestamp":1761371803000},"page":"5-14","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":6,"title":["MotionRefineNet: Fine-Grained Pose Sequence Smoothing and Refinement"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-3829-2849","authenticated-orcid":false,"given":"Haolun","family":"Li","sequence":"first","affiliation":[{"name":"Nanjing University of Posts and Telecommunications, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9532-7633","authenticated-orcid":false,"given":"Weihuang","family":"Liu","sequence":"additional","affiliation":[{"name":"University of Macau, Macao, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-8268-2577","authenticated-orcid":false,"given":"Jiateng","family":"Liu","sequence":"additional","affiliation":[{"name":"Nanjing University of Posts and Telecommunications, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9189-4983","authenticated-orcid":false,"given":"Zhenhua","family":"Tang","sequence":"additional","affiliation":[{"name":"University of Macau, Macau, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1788-3746","authenticated-orcid":false,"given":"Chi-Man","family":"Pun","sequence":"additional","affiliation":[{"name":"University of Macau, Macau, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2872-388X","authenticated-orcid":false,"given":"Qiguang","family":"Miao","sequence":"additional","affiliation":[{"name":"Xidian University, Xi'an, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0953-1057","authenticated-orcid":false,"given":"Feng","family":"Xu","sequence":"additional","affiliation":[{"name":"Tsinghua University, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0148-3713","authenticated-orcid":false,"given":"Hao","family":"Gao","sequence":"additional","affiliation":[{"name":"Nanjing University of Posts and Telecommunications, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"LensNet: An End-to-End Learning Framework for Empirical Point Spread Function Modeling and Lensless Imaging Reconstruction. arXiv","author":"Bai Jiesong","year":"2025","unstructured":"Jiesong Bai, Yuhao Yin, Yihang Dong, Xiaofeng Zhang, Chi-Man Pun, and Xuhang Chen. 2025. LensNet: An End-to-End Learning Framework for Empirical Point Spread Function Modeling and Lensless Imaging Reconstruction. arXiv (2025)."},{"key":"e_1_3_2_1_2_1","volume-title":"An empirical evaluation of generic convolutional and recurrent networks for sequence modeling. arXiv preprint arXiv:1803.01271","author":"Bai Shaojie","year":"2018","unstructured":"Shaojie Bai, J Zico Kolter, and Vladlen Koltun. 2018. An empirical evaluation of generic convolutional and recurrent networks for sequence modeling. arXiv preprint arXiv:1803.01271 (2018)."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/2207676.2208639"},{"key":"e_1_3_2_1_4_1","first-page":"4060","article-title":"Wavelet-Decoupling Contrastive Enhancement Network for Fine-Grained Skeleton-Based Action Recognition. In ICASSP 2024-2024 IEEE International Conference on Acoustics","author":"Chang Haochen","year":"2024","unstructured":"Haochen Chang, Jing Chen, Yilin Li, Jixiang Chen, and Xiaofeng Zhang. 2024. Wavelet-Decoupling Contrastive Enhancement Network for Fine-Grained Skeleton-Based Action Recognition. In ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). 4060-4064.","journal-title":"Speech and Signal Processing (ICASSP)."},{"key":"e_1_3_2_1_5_1","volume-title":"High-Fidelity Functional Ultrasound Reconstruction via A Visual Auto-Regressive Framework. arxiv","author":"Chen Xuhang","year":"2025","unstructured":"Xuhang Chen, Zhuo Li, Yanyan Shen, Mufti Mahmud, Hieu Pham, Chi-Man Pun, and Shuqiang Wang. 2025a. High-Fidelity Functional Ultrasound Reconstruction via A Visual Auto-Regressive Framework. arxiv (2025)."},{"key":"e_1_3_2_1_6_1","volume-title":"Kim-Fung Tsang, Chi-Man Pun, and Shuqiang Wang.","author":"Chen Xuhang","year":"2025","unstructured":"Xuhang Chen, Michael Kwok-Po Ng, Kim-Fung Tsang, Chi-Man Pun, and Shuqiang Wang. 2025b. ConnectomeDiffuser: Generative AI Enables Brain Network Construction from Diffusion Tensor Imaging. arxiv (2025)."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00742"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00756"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/D14-1179"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00200"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2023.109806"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00624"},{"key":"e_1_3_2_1_13_1","volume-title":"Asynchronous Joint-based Temporal Pooling for Skeleton-based Action Recognition","author":"Gunasekara Shanaka Ramesh","year":"2024","unstructured":"Shanaka Ramesh Gunasekara, Wanqing Li, Jack Yang, and Philip Ogunbona. 2024. Asynchronous Joint-based Temporal Pooling for Skeleton-based Action Recognition. IEEE Transactions on Circuits and Systems for Video Technology (2024)."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2025.3525593"},{"key":"e_1_3_2_1_15_1","first-page":"1","article-title":"Underwater image restoration via polymorphic large kernel cnns","author":"Guo Xiaojiao","year":"2025","unstructured":"Xiaojiao Guo, Yihang Dong, Xuhang Chen, Weiwen Chen, Zimeng Li, FuChen Zheng, and Chi-Man Pun. 2025b. Underwater image restoration via polymorphic large kernel cnns. In ICASSP. 1-5.","journal-title":"ICASSP."},{"key":"e_1_3_2_1_16_1","volume-title":"An asymmetric calibrated transformer network for underwater image restoration. The Visual Computer","author":"Guo Xiaojiao","year":"2025","unstructured":"Xiaojiao Guo, Shenghong Luo, Yihang Dong, Zexiao Liang, Zimeng Li, Xiujun Zhang, and Xuhang Chen. 2025c. An asymmetric calibrated transformer network for underwater image restoration. The Visual Computer (2025), 1-13."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00430"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"crossref","unstructured":"Daniel Holden Jun Saito Taku Komura and Thomas Joyce. 2015. Learning motion manifolds with convolutional autoencoders. In SIGGRAPH Asia 2015 technical briefs. 1-4.","DOI":"10.1145\/2820903.2820918"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2019.2953325"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10447880"},{"key":"e_1_3_2_1_21_1","volume-title":"6m: Large scale datasets and predictive methods for 3d human sensing in natural environments","author":"Ionescu Catalin","year":"2013","unstructured":"Catalin Ionescu, Dragos Papava, Vlad Olaru, and Cristian Sminchisescu. 2013. Human3. 6m: Large scale datasets and predictive methods for 3d human sensing in natural environments. IEEE transactions on pattern analysis and machine intelligence, Vol. 36, 7 (2013), 1325-1339."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/3DV53792.2021.00015"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00530"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01094"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00234"},{"key":"e_1_3_2_1_26_1","volume-title":"Proceedings of the Asian conference on computer vision.","author":"Lebailly Tim","year":"2020","unstructured":"Tim Lebailly, Sena Kiciroglu, Mathieu Salzmann, Pascal Fua, and Wei Wang. 2020. Motion prediction using temporal inception module. In Proceedings of the Asian conference on computer vision."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/BIBM58861.2023.10385700"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2022.3180737"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i1.25214"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49660.2025.10890436"},{"key":"e_1_3_2_1_31_1","first-page":"7614","article-title":"High-Fidelity Document Stain Removal via A Large-Scale Real-World Dataset and A Memory-Augmented Transformer","author":"Li Mingxian","year":"2025","unstructured":"Mingxian Li, Hao Sun, Yingtie Lei, Xiaofeng Zhang, Yihang Dong, Yilin Zhou, Zimeng Li, and Xuhang Chen. 2025a. High-Fidelity Document Stain Removal via A Large-Scale Real-World Dataset and A Memory-Augmented Transformer. In WACV. 7614-7624.","journal-title":"WACV."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01315"},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00064"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3588432.3591490"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612368"},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00999"},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/BIBM62325.2024.10822311"},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/2816795.2818013"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00059"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00958"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.288"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV57701.2024.00677"},{"key":"e_1_3_2_1_43_1","volume-title":"Stacked Hourglass Networks for Human Pose Estimation. In European Conference on Computer Vision. 483-499","author":"Newell Alejandro","year":"2016","unstructured":"Alejandro Newell, Kaiyu Yang, and Jia Deng. 2016. Stacked Hourglass Networks for Human Pose Estimation. In European Conference on Computer Vision. 483-499."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW63382.2024.00596"},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1063\/1.4822961"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00858"},{"key":"e_1_3_2_1_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00202"},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1145\/3528223.3530178"},{"key":"e_1_3_2_1_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00584"},{"key":"e_1_3_2_1_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00464"},{"key":"e_1_3_2_1_51_1","volume-title":"Long-term temporal convolutions for action recognition","author":"Varol G\u00fcl","year":"2017","unstructured":"G\u00fcl Varol, Ivan Laptev, and Cordelia Schmid. 2017. Long-term temporal convolutions for action recognition. IEEE transactions on pattern analysis and machine intelligence, Vol. 40, 6 (2017), 1510-1517."},{"key":"e_1_3_2_1_52_1","volume-title":"Attention is all you need. NeurIPS","author":"Vaswani A","year":"2017","unstructured":"A Vaswani. 2017. Attention is all you need. NeurIPS (2017)."},{"key":"e_1_3_2_1_53_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01249-6_37"},{"key":"e_1_3_2_1_54_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547780"},{"key":"e_1_3_2_1_55_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00544"},{"key":"e_1_3_2_1_56_1","volume-title":"Exploiting Temporal Correlations for 3D Human Pose Estimation","author":"Wang Ruibin","year":"2023","unstructured":"Ruibin Wang, Xianghua Ying, and Bowei Xing. 2023c. Exploiting Temporal Correlations for 3D Human Pose Estimation. IEEE Transactions on Multimedia (2023)."},{"key":"e_1_3_2_1_57_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00179"},{"key":"e_1_3_2_1_58_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00363"},{"key":"e_1_3_2_1_59_1","volume-title":"Himorenet: A hierarchical model for human motion refinement","author":"Wang Zhiming","year":"2023","unstructured":"Zhiming Wang, Juan Wang, Ning Ge, and Jianhua Lu. 2023b. Himorenet: A hierarchical model for human motion refinement. IEEE Signal Processing Letters (2023)."},{"key":"e_1_3_2_1_60_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01286"},{"key":"e_1_3_2_1_61_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681179"},{"key":"e_1_3_2_1_62_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3681009"},{"key":"e_1_3_2_1_63_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00099"},{"key":"e_1_3_2_1_64_1","volume-title":"Vitpose: Simple vision transformer baselines for human pose estimation. Advances in neural information processing systems","author":"Xu Yufei","year":"2022","unstructured":"Yufei Xu, Jing Zhang, Qiming Zhang, and Dacheng Tao. 2022. Vitpose: Simple vision transformer baselines for human pose estimation. Advances in neural information processing systems, Vol. 35 (2022), 38571-38584."},{"key":"e_1_3_2_1_65_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v32i1.12328"},{"key":"e_1_3_2_1_66_1","volume-title":"Hierarchical local temporal network for 2d-to-3d human pose estimation","author":"Yan Xin","year":"2024","unstructured":"Xin Yan, Jiucheng Xie, Mengqi Liu, Haolun Li, and Hao Gao. 2024. Hierarchical local temporal network for 2d-to-3d human pose estimation. IEEE Internet of Things Journal (2024)."},{"key":"e_1_3_2_1_67_1","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2024.3392686"},{"key":"e_1_3_2_1_68_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01374"},{"key":"e_1_3_2_1_69_1","volume-title":"Recursive implementation of the Gaussian filter. Signal processing","author":"Young Ian T","year":"1995","unstructured":"Ian T Young and Lucas J Van Vliet. 1995. Recursive implementation of the Gaussian filter. Signal processing, Vol. 44, 2 (1995), 139-151."},{"key":"e_1_3_2_1_70_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20065-6_36"},{"key":"e_1_3_2_1_71_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3547773"},{"key":"e_1_3_2_1_72_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3680880"},{"key":"e_1_3_2_1_73_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01288"},{"key":"e_1_3_2_1_74_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00857"},{"key":"e_1_3_2_1_75_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00200"},{"key":"e_1_3_2_1_76_1","first-page":"2468","article-title":"DocDeshadower","author":"Zhou Ziyang","year":"2024","unstructured":"Ziyang Zhou, Yingtie Lei, Xuhang Chen, Shenghong Luo, Wenjun Zhang, Chi-Man Pun, and Zhen Wang. 2024a. DocDeshadower: Frequency-Aware Transformer for Document Shadow Removal. In SMC. 2468-2473.","journal-title":"Frequency-Aware Transformer for Document Shadow Removal. In SMC."},{"key":"e_1_3_2_1_77_1","volume-title":"Vision mamba: Efficient visual representation learning with bidirectional state space model. arXiv preprint arXiv:2401.09417","author":"Zhu Lianghui","year":"2024","unstructured":"Lianghui Zhu, Bencheng Liao, Qian Zhang, Xinlong Wang, Wenyu Liu, and Xinggang Wang. 2024. Vision mamba: Efficient visual representation learning with bidirectional state space model. arXiv preprint arXiv:2401.09417 (2024)."},{"key":"e_1_3_2_1_78_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01385"},{"key":"e_1_3_2_1_79_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01128"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3754506","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T04:03:22Z","timestamp":1765339402000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3754506"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":79,"alternative-id":["10.1145\/3746027.3754506","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3754506","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}