{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,7]],"date-time":"2025-06-07T06:10:10Z","timestamp":1749276610706,"version":"3.41.0"},"publisher-location":"Singapore","reference-count":44,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819665983","type":"print"},{"value":"9789819665969","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-981-96-6596-9_13","type":"book-chapter","created":{"date-parts":[[2025,6,7]],"date-time":"2025-06-07T05:31:22Z","timestamp":1749274282000},"page":"180-194","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["FOPS-V: Feature-Aware Optimization and\u00a0Parallel Scale Fusion for\u00a03D Human Reconstruction in\u00a0Video"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-5771-224X","authenticated-orcid":false,"given":"Yang","family":"Huang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3640-3229","authenticated-orcid":false,"given":"Guoheng","family":"Huang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8213-041X","authenticated-orcid":false,"given":"Lianglun","family":"Cheng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-1550-5121","authenticated-orcid":false,"given":"Yejing","family":"Huo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6000-3914","authenticated-orcid":false,"given":"Xuhang","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7490-6695","authenticated-orcid":false,"given":"Xiaochen","family":"Yuan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6428-5645","authenticated-orcid":false,"given":"Guo","family":"Zhong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1788-3746","authenticated-orcid":false,"given":"Chi-Man","family":"Pun","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,6,8]]},"reference":[{"key":"13_CR1","doi-asserted-by":"crossref","unstructured":"Arnab, A., Doersch, C., Zisserman, A.: Exploiting temporal context for 3D human pose estimation in the wild. In: CVPR, pp. 3390\u20133399 (2019)","DOI":"10.1109\/CVPR.2019.00351"},{"key":"13_CR2","doi-asserted-by":"crossref","unstructured":"Chatzis, T., Konstantinidis, D., Dimitropoulos, K., Daras, P.: 3D pose estimation using a global and local cross-attention mechanism. In: IST, pp.\u00a01\u20136 (2023)","DOI":"10.1109\/IST59124.2023.10355655"},{"key":"13_CR3","unstructured":"Chen, H., He, B., Wang, H., Ren, Y., Lim, S.N., Shrivastava, A.: Nerv: neural representations for videos. arXiv (2021)"},{"key":"13_CR4","unstructured":"Chen, P., et al.: Pathformer: multi-scale transformers with adaptive pathways for time series forecasting. ArXiv (2024)"},{"key":"13_CR5","doi-asserted-by":"crossref","unstructured":"Chen, X., Lei, B., Pun, C.M., Wang, S.: Brain diffuser: an end-to-end brain image to brain network pipeline. In: PRCV,pp. 16\u201326 (2023)","DOI":"10.1007\/978-981-99-8558-6_2"},{"key":"13_CR6","doi-asserted-by":"crossref","unstructured":"Chen, X., Pun, C.M., Wang, S.: Medprompt: cross-modal prompting for multi-task medical image translation. arXiv (2023)","DOI":"10.1007\/978-981-97-8496-7_5"},{"key":"13_CR7","doi-asserted-by":"crossref","unstructured":"Chen, Y., Wang, Z., Peng, Y., Zhang, Z., Yu, G., Sun, J.: Cascaded pyramid network for multi-person pose estimation. In: CVPR, pp. 7103\u20137112 (2017)","DOI":"10.1109\/CVPR.2018.00742"},{"key":"13_CR8","doi-asserted-by":"crossref","unstructured":"Choi, H., Moon, G., Lee, K.M.: Beyond static features for temporally consistent 3D human pose and shape from a video. In: CVPR, pp. 1964\u20131973 (2020)","DOI":"10.1109\/CVPR46437.2021.00200"},{"key":"13_CR9","doi-asserted-by":"crossref","unstructured":"Guo, X., Chen, X., Luo, S., Wang, S., Pun, C.M.: Dual-hybrid attention network for specular highlight removal. In: ACM MM (2024)","DOI":"10.1145\/3664647.3680745"},{"key":"13_CR10","doi-asserted-by":"crossref","unstructured":"Hassanin, M., Khamiss, A., Bennamoun, Boussaid, F., Radwan, I.: Crossformer: cross spatio-temporal transformer for 3D human pose estimation. ArXiv (2022)","DOI":"10.2139\/ssrn.4213439"},{"key":"13_CR11","doi-asserted-by":"crossref","unstructured":"Hu, X., Zhang, Z., Jiang, Z., Chaudhuri, S., Yang, Z., Nevatia, R.: Span: spatial pyramid attention network forimage manipulation localization. In: ECCV (2020)","DOI":"10.1007\/978-3-030-58589-1_19"},{"key":"13_CR12","doi-asserted-by":"crossref","unstructured":"Huang, X., Belongie, S.J.: Arbitrary style transfer in real-time with adaptive instance normalization. In: ICCV, pp. 1510\u20131519 (2017)","DOI":"10.1109\/ICCV.2017.167"},{"key":"13_CR13","doi-asserted-by":"crossref","unstructured":"Ionescu, C., Papava, D., Olaru, V., Sminchisescu, C.: Human3.6M: large scale datasets and predictive methods for 3D human sensing in natural environments. TPAMI 36, 1325\u20131339 (2014)","DOI":"10.1109\/TPAMI.2013.248"},{"key":"13_CR14","doi-asserted-by":"crossref","unstructured":"Jiang, Y., Chen, X., Pun, C.M., Wang, S., Feng, W.: MFDNet: multi-frequency deflare network for efficient nighttime flare removal. Vis. Comput. 1\u201314 (2024)","DOI":"10.1007\/s00371-024-03540-x"},{"key":"13_CR15","doi-asserted-by":"crossref","unstructured":"Kanazawa, A., Zhang, J.Y., Felsen, P., Malik, J.: Learning 3D human dynamics from video. In: CVPR, pp. 5607\u20135616 (2018)","DOI":"10.1109\/CVPR.2019.00576"},{"key":"13_CR16","doi-asserted-by":"crossref","unstructured":"Kocabas, M., Athanasiou, N., Black, M.J.: Vibe: video inference for human body pose and shape estimation. In: CVPR, pp. 5252\u20135262 (2019)","DOI":"10.1109\/CVPR42600.2020.00530"},{"key":"13_CR17","doi-asserted-by":"crossref","unstructured":"Kolotouros, N., Pavlakos, G., Black, M.J., Daniilidis, K.: Learning to reconstruct 3D human pose and shape via model-fitting in the loop. In: ICCV, pp. 2252\u20132261 (2019)","DOI":"10.1109\/ICCV.2019.00234"},{"key":"13_CR18","doi-asserted-by":"crossref","unstructured":"Li, J., Xu, C., Chen, Z., Bian, S., Yang, L., Lu, C.: Hybrik: a hybrid analytical-neural inverse kinematics solution for 3D human pose and shape estimation. In: CVPR, pp. 3382\u20133392 (2020)","DOI":"10.1109\/CVPR46437.2021.00339"},{"issue":"2","key":"13_CR19","doi-asserted-by":"publisher","first-page":"336","DOI":"10.1007\/s11390-024-3414-z","volume":"39","author":"Z Li","year":"2024","unstructured":"Li, Z., Chen, X., Guo, S., Wang, S., Pun, C.M.: Wavenhancer: unifying wavelet and transformer for image enhancement. J. Comput. Sci. Technol. 39(2), 336\u2013345 (2024)","journal-title":"J. Comput. Sci. Technol."},{"key":"13_CR20","doi-asserted-by":"crossref","unstructured":"Loper, M., Mahmood, N., Romero, J., Pons-Moll, G., Black, M.J.: SMPL: a skinned multi-person linear model. In: Seminal Graphics Papers: Pushing the Boundaries, Vol. 2 (2023)","DOI":"10.1145\/3596711.3596800"},{"key":"13_CR21","doi-asserted-by":"crossref","unstructured":"Luo, Z., Golestaneh, S.A., Kitani, K.M.: 3D human motion estimation via motion compression and refinement. In: ACCV (2020)","DOI":"10.1007\/978-3-030-69541-5_20"},{"key":"13_CR22","doi-asserted-by":"crossref","unstructured":"Ma, T., Liu, Z., Si, Y., Fu, C.: PSA-YOLO: license plate detection method based on pyramid segmentation attention in complex scenes. In: ICIVC, pp. 238\u2013243 (2022)","DOI":"10.1109\/ICIVC55077.2022.9886741"},{"key":"13_CR23","doi-asserted-by":"crossref","unstructured":"von Marcard, T., Henschel, R., Black, M.J., Rosenhahn, B., Pons-Moll, G.: Supplementary material to: recovering accurate 3D human pose in the wild using IMUs and a moving camera (2018)","DOI":"10.1007\/978-3-030-01249-6_37"},{"key":"13_CR24","doi-asserted-by":"crossref","unstructured":"Mehta, D., et al.: Monocular 3D human pose estimation in the wild using improved CNN supervision. In: 3DV, pp. 506\u2013516 (2016)","DOI":"10.1109\/3DV.2017.00064"},{"key":"13_CR25","doi-asserted-by":"crossref","unstructured":"Saito, S., Simon, T., Saragih, J.M., Joo, H.: PIFuHD: multi-level pixel-aligned implicit function for high-resolution 3D human digitization. In: CVPR, pp. 81\u201390 (2020)","DOI":"10.1109\/CVPR42600.2020.00016"},{"key":"13_CR26","doi-asserted-by":"crossref","unstructured":"Seabe, P.L., Moutsinga, C.R.B., Pindza, E.: Forecasting cryptocurrency prices using LSTM, GRU, and bi-directional LSTM: a deep learning approach. Fractal Fractional (2023)","DOI":"10.3390\/fractalfract7020203"},{"key":"13_CR27","doi-asserted-by":"crossref","unstructured":"Sun, X., Xiao, B., Liang, S., Wei, Y.: Integral human pose regression. In: ECCV (2017)","DOI":"10.1007\/978-3-030-01231-1_33"},{"key":"13_CR28","doi-asserted-by":"crossref","unstructured":"Sun, Y., Ye, Y., Liu, W., Gao, W., Fu, Y., Mei, T.: Human mesh recovery from monocular images via a skeleton-disentangled representation. In: ICCV, pp. 5348\u20135357 (2019)","DOI":"10.1109\/ICCV.2019.00545"},{"key":"13_CR29","doi-asserted-by":"crossref","unstructured":"Tang, Z., Qiu, Z., Hao, Y., Hong, R., Yao, T.: 3D human pose estimation with spatio-temporal criss-cross attention. In: CVPR, pp. 4790\u20134799 (2023)","DOI":"10.1109\/CVPR52729.2023.00464"},{"key":"13_CR30","unstructured":"Tung, H.Y.F., Tung, H.W., Yumer, E., Fragkiadaki, K.: Self-supervised learning of motion capture. In: NeurIPS (2017)"},{"key":"13_CR31","unstructured":"Vaswani, A., et al.: Attention is all you need. In: NeurIPS (2017)"},{"key":"13_CR32","doi-asserted-by":"crossref","unstructured":"Wan, Z., Li, Z., Tian, M., Liu, J., Yi, S., Li, H.: Encoder-decoder with multi-level attention for 3D human shape and pose estimation. In: ICCV, pp. 13013\u201313022 (2021)","DOI":"10.1109\/ICCV48922.2021.01279"},{"key":"13_CR33","doi-asserted-by":"crossref","unstructured":"Wang, T., Liu, H., Ding, R., Li, W., You, Y., Li, X.: Interweaved graph and attention network for 3D human pose estimation. In: ICASSP, pp.\u00a01\u20135 (2023)","DOI":"10.1109\/ICASSP49357.2023.10097259"},{"key":"13_CR34","doi-asserted-by":"crossref","unstructured":"Wei, W.L., Lin, J.C., Liu, T.L., Liao, H.Y.M.: Capturing humans in motion: temporal-attentive 3D human pose and shape estimation from monocular video. In: CVPR, pp. 13201\u201313210 (2022)","DOI":"10.1109\/CVPR52688.2022.01286"},{"key":"13_CR35","doi-asserted-by":"crossref","unstructured":"Wu, C., Xiao, Y., Zhang, B., Zhang, M., Cao, Z., Zhou, J.T.: C3P: cross-domain pose prior propagation for weakly supervised 3D human pose estimation. In: ECCV (2022)","DOI":"10.1007\/978-3-031-20065-6_32"},{"key":"13_CR36","unstructured":"Xu, J., Sun, X., Zhang, Z., Zhao, G., Lin, J.: Understanding and improving layer normalization. ArXiv (2019)"},{"key":"13_CR37","unstructured":"Xu, M., et al.: Spatial-temporal transformer networks for traffic flow forecasting. ArXiv (2020)"},{"key":"13_CR38","doi-asserted-by":"crossref","unstructured":"You, Y., Liu, H., Wang, T., Li, W., Ding, R., Li, X.: Co-evolution of pose and mesh for 3D human body estimation from video. In: ICCV, pp. 14917\u201314927 (2023)","DOI":"10.1109\/ICCV51070.2023.01374"},{"key":"13_CR39","doi-asserted-by":"crossref","unstructured":"Yuan, L., et al.: Tokens-to-token ViT: training vision transformers from scratch on ImageNet. In: ICCV, pp. 538\u2013547 (2021)","DOI":"10.1109\/ICCV48922.2021.00060"},{"key":"13_CR40","unstructured":"Zhang, H., Zu, K., Lu, J., Zou, Y., Meng, D.: EPSANet: an efficient pyramid squeeze attention block on convolutional neural network. In: ACCV (2021)"},{"key":"13_CR41","doi-asserted-by":"crossref","unstructured":"Zheng, F., et al.: Smaformer: synergistic multi-attention transformer for medical image segmentation. arXiv (2024)","DOI":"10.1109\/BIBM62325.2024.10822736"},{"key":"13_CR42","doi-asserted-by":"crossref","unstructured":"Zhi, T., Lassner, C., Tung, T., Stoll, C., Narasimhan, S.G., Vo, M.: Texmesh: reconstructing detailed human texture and geometry from RGB-D video. In: ECCV (2020)","DOI":"10.1007\/978-3-030-58607-2_29"},{"key":"13_CR43","doi-asserted-by":"crossref","unstructured":"Zhou, Z., et al.: DocDeshadower: frequency-aware transformer for document shadow removal. arXiv (2024)","DOI":"10.1109\/SMC54092.2024.10831480"},{"key":"13_CR44","doi-asserted-by":"crossref","unstructured":"Zhu, P., Abdal, R., Qin, Y., Wonka, P.: Sean: image synthesis with semantic region-adaptive normalization. In: CVPR, pp. 5103\u20135112 (2019)","DOI":"10.1109\/CVPR42600.2020.00515"}],"container-title":["Lecture Notes in Computer Science","Neural Information Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-96-6596-9_13","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,7]],"date-time":"2025-06-07T05:31:33Z","timestamp":1749274293000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-96-6596-9_13"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9789819665983","9789819665969"],"references-count":44,"URL":"https:\/\/doi.org\/10.1007\/978-981-96-6596-9_13","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"8 June 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICONIP","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Neural Information Processing","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Auckland","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"New Zealand","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2 December 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"6 December 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"31","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"iconip2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/iconip2024.org","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}