{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T03:04:44Z","timestamp":1740107084877,"version":"3.37.3"},"reference-count":44,"publisher":"Springer Science and Business Media LLC","issue":"8","license":[{"start":{"date-parts":[[2023,6,24]],"date-time":"2023-06-24T00:00:00Z","timestamp":1687564800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,6,24]],"date-time":"2023-06-24T00:00:00Z","timestamp":1687564800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["61702335"],"award-info":[{"award-number":["61702335"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"National Science Foundation of Guangdong Province of China","award":["2021A1515011632"],"award-info":[{"award-number":["2021A1515011632"]}]},{"DOI":"10.13039\/501100009019","name":"Shenzhen University","doi-asserted-by":"crossref","award":["001203234"],"award-info":[{"award-number":["001203234"]}],"id":[{"id":"10.13039\/501100009019","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Vis Comput"],"published-print":{"date-parts":[[2023,8]]},"DOI":"10.1007\/s00371-023-02959-y","type":"journal-article","created":{"date-parts":[[2023,6,24]],"date-time":"2023-06-24T17:02:36Z","timestamp":1687626156000},"page":"3247-3257","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Deformable patch embedding-based shift module-enhanced transformer for panoramic action recognition"],"prefix":"10.1007","volume":"39","author":[{"given":"Xiaoyan","family":"Zhang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yujie","family":"Cui","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yongkai","family":"Huo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,6,24]]},"reference":[{"key":"2959_CR1","doi-asserted-by":"crossref","unstructured":"Lee, Y., Jeong, J., Yun, J., Cho, W., Yoon, K.-J.: SpherePHD: applying CNNs on a spherical polyhedron representation of 360$$^\\circ $$ Images. In: 2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognitions, pp.\u00a09173\u20139181. (2019)","DOI":"10.1109\/CVPR.2019.00940"},{"key":"2959_CR2","doi-asserted-by":"crossref","unstructured":"Li, J., Liu, J., Wong, Y., Nishimura, S., Kankanhalli, M.S.: Weakly-supervised multi-person action recognition in 360$$^{\\circ }$$ videos. In: 2020 IEEE Winter Conference on Applications of Computer Vision, pp.\u00a0497\u2013505. (2020)","DOI":"10.1109\/WACV45572.2020.9093283"},{"key":"2959_CR3","doi-asserted-by":"crossref","unstructured":"Carreira, J., Zisserman, A.: Quo vadis, action recognition? A new model and the kinetics dataset. In: 2017 IEEE Conference on Computer Vision and Pattern Recognition, pp.\u00a04724\u20134733. (2017)","DOI":"10.1109\/CVPR.2017.502"},{"key":"2959_CR4","doi-asserted-by":"crossref","unstructured":"Li, D., Shi, W.: Partially occluded skeleton action recognition based on multi-stream fusion graph convolutional networks. In: Advances in Computer Graphics - 38th Computer Graphics International Conference, 2021, vol.\u00a013002 of Lecture Notes in Computer Science, pp.\u00a0178\u2013189, Springer, (2021)","DOI":"10.1007\/978-3-030-89029-2_14"},{"key":"2959_CR5","doi-asserted-by":"crossref","unstructured":"Lin, J., Gan, C., Han, S.: TSM: temporal shift module for efficient video understanding. In: 2019 IEEE\/CVF International Conference on Computer Vision, pp.\u00a07083\u20137092, (2019)","DOI":"10.1109\/ICCV.2019.00718"},{"key":"2959_CR6","volume-title":"Flattening the Earth: Two Thousand Years of Map Projections","author":"JP Snyder","year":"1993","unstructured":"Snyder, J.P.: Flattening the Earth: Two Thousand Years of Map Projections. University of Chicago Press, Chicago, USA (1993)"},{"key":"2959_CR7","doi-asserted-by":"publisher","first-page":"26","DOI":"10.1016\/j.image.2018.05.005","volume":"69","author":"R Monroy","year":"2017","unstructured":"Monroy, R., Lutz, S., Chalasani, T., Smolic, A.: SalNet360: saliency maps for omni-directional images with CNN. Signal Process. Image Commun. 69, 26\u201334 (2017)","journal-title":"Signal Process. Image Commun."},{"key":"2959_CR8","doi-asserted-by":"crossref","unstructured":"Eder, M., Shvets, M., Lim, J., Frahm, J.-M.: Tangent images for mitigating spherical distortion. In: 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.\u00a012423\u201312431, (2020)","DOI":"10.1109\/CVPR42600.2020.01244"},{"key":"2959_CR9","doi-asserted-by":"crossref","unstructured":"Li, Y., Barnes, C., Huang, K., Zhang, F.: Deep 360$$^{\\circ }$$ optical flow estimation based on multi-projection fusion. In: European Conference on Computer Vision, pp.\u00a0336\u2013352, (2022)","DOI":"10.1007\/978-3-031-19833-5_20"},{"key":"2959_CR10","doi-asserted-by":"crossref","unstructured":"Dai, J., Qi, H., Xiong, Y., Li, Y., Zhang, G., Hu, H., Wei, Y.: Deformable convolutional networks. In: 2017 IEEE International Conference on Computer Vision, pp.\u00a0764\u2013773, (2017)","DOI":"10.1109\/ICCV.2017.89"},{"key":"2959_CR11","doi-asserted-by":"crossref","unstructured":"Bhandari, K., DeLaGarza, M.A., Zong, Z., Latapie, H., Yan, Y.: EGOK360: a 360 egocentric kinetic human activity video dataset. In: 2020 IEEE International Conference on Image Processing, pp.\u00a0266\u2013270, (2020)","DOI":"10.1109\/ICIP40778.2020.9191256"},{"key":"2959_CR12","first-page":"568","volume":"27","author":"K Simonyan","year":"2014","unstructured":"Simonyan, K., Zisserman, A.: Two-stream convolutional networks for action recognition in videos. Adv. Neural Inf. Process. Syst. 27, 568\u2013576 (2014)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"2959_CR13","doi-asserted-by":"publisher","first-page":"2740","DOI":"10.1109\/TPAMI.2018.2868668","volume":"41","author":"L Wang","year":"2019","unstructured":"Wang, L., Xiong, Y., Wang, Z., Qiao, Y., Lin, D., Tang, X., Gool, L.V.: Temporal segment networks for action recognition in videos. IEEE Trans. Pattern Anal. Mach. Intell. 41, 2740\u20132755 (2019)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"2959_CR14","doi-asserted-by":"publisher","first-page":"2799","DOI":"10.1109\/TIP.2018.2890749","volume":"28","author":"Z Tu","year":"2019","unstructured":"Tu, Z., Li, H., Zhang, D., Dauwels, J., Li, B., Yuan, J.: Action-stage emphasized spatiotemporal VLAD for video action recognition. IEEE Trans. Image Process. 28, 2799\u20132812 (2019)","journal-title":"IEEE Trans. Image Process."},{"key":"2959_CR15","doi-asserted-by":"publisher","first-page":"8429","DOI":"10.1109\/TIP.2020.3013168","volume":"29","author":"L Tian","year":"2020","unstructured":"Tian, L., Tu, Z., Zhang, D., Liu, J., Li, B., Yuan, J.: Unsupervised learning of optical flow with CNN-based non-local filtering. IEEE Trans. Image Process. 29, 8429\u20138442 (2020)","journal-title":"IEEE Trans. Image Process."},{"key":"2959_CR16","doi-asserted-by":"crossref","unstructured":"Sudhakaran, S., Escalera, S., Lanz, O.: LSTA: long short-term attention for egocentric action recognition. In: 2019 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.\u00a09946\u20139955, (2019)","DOI":"10.1109\/CVPR.2019.01019"},{"key":"2959_CR17","unstructured":"Gers, F.A., Schmidhuber, J., Cummins, F.: Learning to forget: continual prediction with LSTM. In: 1999 Ninth International Conference on Artificial Neural Networks, vol.\u00a02, pp.\u00a0850\u2013855, (2000)"},{"key":"2959_CR18","doi-asserted-by":"crossref","unstructured":"Zhou, B., Andonian, A., Oliva, A., Torralba, A.: Temporal relational reasoning in videos. In: European Conference on Computer Vision, pp.\u00a0831\u2013846, (2018)","DOI":"10.1007\/978-3-030-01246-5_49"},{"key":"2959_CR19","doi-asserted-by":"crossref","unstructured":"Wang, L., Tong, Z., Ji, B., Wu, G.: TDN: temporal difference networks for efficient action recognition. In: 2021 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.\u00a01895\u20131904, (2021)","DOI":"10.1109\/CVPR46437.2021.00193"},{"key":"2959_CR20","doi-asserted-by":"crossref","unstructured":"Yan, F., Wen, J., Li, Z., Zhou, Z.: Monocular dense SLAM with consistent deep depth prediction. In: Advances in Computer Graphics-38th Computer Graphics International Conference, 2021, vol.\u00a013002 of Lecture Notes in Computer Science, pp.\u00a0113\u2013124, (2021)","DOI":"10.1007\/978-3-030-89029-2_9"},{"key":"2959_CR21","doi-asserted-by":"crossref","unstructured":"Zhang, H., Guo, M., Zhao, W., Huang, J., Meng, Z., Lu, P., Sen, L., Sheng, B.: Visual indoor navigation using mobile augmented reality. In: Advances in Computer Graphics - 39th Computer Graphics International Conference, 2022, Virtual Event, Sept 12\u201316, 2022, Proceedings, vol.\u00a013443 of Lecture Notes in Computer Science, pp.\u00a0145\u2013156, (2022)","DOI":"10.1007\/978-3-031-23473-6_12"},{"key":"2959_CR22","unstructured":"Jiang, C.M., Huang, J., Kashinath Prabhat, K., Marcus, P., Nie\u00dfner, M.: Spherical CNNs on unstructured grids. In: International Conference on Learning Representations, (2019)"},{"key":"2959_CR23","doi-asserted-by":"crossref","unstructured":"Han, R., Yan, H., Li, J., Wang, S., Feng, W., Wang, S.: Panoramic human activity recognition. In: European Conference on Computer Vision, p.\u00a0244-261, (2022)","DOI":"10.1007\/978-3-031-19772-7_15"},{"key":"2959_CR24","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Weissenborn, D., Zhai, X., Unterthiner, T., Dehghani, M., Minderer, M., Heigold, G., Gelly, S., Uszkoreit, J., Houlsby, N.: An image is worth 16 $$\\times $$ 16 words: transformers for image recognition at scale. In: International Conference on Learning Representations, (2021)"},{"key":"2959_CR25","doi-asserted-by":"crossref","unstructured":"Touvron, H., Cord, M., Sablayrolles, A., Synnaeve, G., J\u00e9gou, H.: Going deeper with image transformers. In Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp.\u00a032\u201342, (2021)","DOI":"10.1109\/ICCV48922.2021.00010"},{"key":"2959_CR26","unstructured":"Touvron, H., Cord, M., Douze, M., Massa, F., Sablayrolles, A., J\u00e9gou, H.: Training data-efficient image transformers & distillation through attention. In: International Conference on Machine Learning, pp.\u00a010347\u201310357, PMLR, (2021)"},{"key":"2959_CR27","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lin, Y., Cao, Y., Hu, H., Wei, Y., Zhang, Z., Lin, S., Guo, B.: Swin transformer: hierarchical vision transformer using shifted windows. In: 2021 IEEE\/CVF International Conference on Computer Vision, pp.\u00a09992\u201310002, IEEE Computer Society, (2021)","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"2959_CR28","unstructured":"Xie, E., Wang, W., Yu, Z., Anandkumar, A., Alvarez, J.M., Luo, P.: SegFormer: simple and efficient design for semantic segmentation with transformers. In: Advances in Neural Information Processing Systems, (2021)"},{"key":"2959_CR29","doi-asserted-by":"crossref","unstructured":"Zhao, H., Jiang, L., Jia, J., Torr, P.H., Koltun, V..: Point transformer. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp.\u00a016259\u201316268, (2021)","DOI":"10.1109\/ICCV48922.2021.01595"},{"key":"2959_CR30","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Li, X., Liu, C., Shuai, B., Zhu, Y., Brattoli, B., Chen, H., Marsic, I., Tighe, J.: VidTr: video transformer without convolutions. In: 2021 IEEE\/CVF International Conference on Computer Vision, pp.\u00a013557\u201313567, (2021)","DOI":"10.1109\/ICCV48922.2021.01332"},{"key":"2959_CR31","doi-asserted-by":"crossref","unstructured":"Chen, Z., Zhu, Y., Zhao, C., Hu, G., Zeng, W., Wang, J., Tang, M.: DPT: deformable patch-based transformer for visual recognition. In: Proceedings of the 29th ACM International Conference on Multimedia, p.\u00a02899-2907, (2021)","DOI":"10.1145\/3474085.3475467"},{"key":"2959_CR32","doi-asserted-by":"crossref","unstructured":"Xia, Z., Pan, X., Song, S., Li, L., Huang, G.: Vision transformer with deformable attention. In: 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.\u00a04784\u20134793, (2022)","DOI":"10.1109\/CVPR52688.2022.00475"},{"key":"2959_CR33","unstructured":"Zhu, X., Su, W., Lu, L., Li, B., Wang, X., Dai, J.: Deformable DETR: deformable transformers for end-to-end object detection. In: International Conference on Learning Representations, (2021)"},{"key":"2959_CR34","doi-asserted-by":"crossref","unstructured":"Yun, H., Lee, S., Kim, G.: Panoramic vision transformer for saliency detection in 360$$^\\circ $$ videos. In: European Conference on Computer Vision, pp.\u00a0422\u2013439, (2022)","DOI":"10.1007\/978-3-031-19833-5_25"},{"key":"2959_CR35","doi-asserted-by":"crossref","unstructured":"Zhang, J., Yang, K., Ma, C., Reiss, S., Peng, K., Stiefelhagen, R.: Bending reality: distortion-aware transformers for adapting to panoramic semantic segmentation. In: 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.\u00a016896\u201316906, (2022)","DOI":"10.1109\/CVPR52688.2022.01641"},{"key":"2959_CR36","first-page":"6000","volume":"30","author":"A Vaswani","year":"2017","unstructured":"Vaswani, A., Shazeer, N., Parmar, N., Uszkoreit, J., Jones, L., Gomez, A.N., Kaiser, L., Polosukhin, I.: Attention is all you need. Adv. Neural Inf. Process. Syst. 30, 6000\u20136010 (2017)","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"2959_CR37","doi-asserted-by":"crossref","unstructured":"Brox, T., Bruhn, A., Papenberg, N., Weickert, J.: High accuracy optical flow estimation based on a theory for warping. In: European Conference on Computer Vision, pp.\u00a025\u201336, (2004)","DOI":"10.1007\/978-3-540-24673-2_3"},{"key":"2959_CR38","unstructured":"Steiner, A., Kolesnikov, A., Zhai, X., Wightman, R., Uszkoreit, J., Beyer, L.: How to train your ViT? Data, augmentation, and regularization in vision transformers. Trans. Mach. Learn. Res., (2022)"},{"key":"2959_CR39","unstructured":"Sharir, G., Noy, A., Zelnik-Manor, L.: An image is worth 16 $$\\times $$ 16 words, what is a video worth?. arXiv preprint arXiv:2103.13915, (2021)"},{"key":"2959_CR40","doi-asserted-by":"crossref","unstructured":"Abnar, S., Zuidema, W.: In: Quantifying Attention Flow in Transformers, pp.\u00a04190\u20134197, (2020)","DOI":"10.18653\/v1\/2020.acl-main.385"},{"key":"2959_CR41","doi-asserted-by":"crossref","unstructured":"Sudhakaran, S., Escalera, S., Lanz, O.: Gate-shift networks for video action recognition. In: 2020 IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.\u00a01099\u20131108, (2020)","DOI":"10.1109\/CVPR42600.2020.00118"},{"key":"2959_CR42","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C., Fan, H., Malik, J., He, K.: SlowFast networks for video recognition. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, (2019)","DOI":"10.1109\/ICCV.2019.00630"},{"key":"2959_CR43","unstructured":"Bertasius, G., Wang, H., Torresani, L.: Is space-time attention all you need for video understanding?. In: Proceedings of the 38th International Conference on Machine Learning, vol.\u00a0139, pp.\u00a0813\u2013824, (2021)"},{"key":"2959_CR44","doi-asserted-by":"crossref","unstructured":"Arnab, A., Dehghani, M., Heigold, G., Sun, C., Lucic, M., Schmid, C.: ViViT: a video vision transformer. In: 2021 IEEE\/CVF International Conference on Computer Vision, pp.\u00a06816\u20136826, (2021)","DOI":"10.1109\/ICCV48922.2021.00676"}],"container-title":["The Visual Computer"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-023-02959-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00371-023-02959-y\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00371-023-02959-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,8,18]],"date-time":"2023-08-18T11:03:26Z","timestamp":1692356606000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00371-023-02959-y"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,6,24]]},"references-count":44,"journal-issue":{"issue":"8","published-print":{"date-parts":[[2023,8]]}},"alternative-id":["2959"],"URL":"https:\/\/doi.org\/10.1007\/s00371-023-02959-y","relation":{},"ISSN":["0178-2789","1432-2315"],"issn-type":[{"type":"print","value":"0178-2789"},{"type":"electronic","value":"1432-2315"}],"subject":[],"published":{"date-parts":[[2023,6,24]]},"assertion":[{"value":"9 June 2023","order":1,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 June 2023","order":2,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no conflicts of interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}