{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,26]],"date-time":"2026-02-26T15:23:55Z","timestamp":1772119435907,"version":"3.50.1"},"reference-count":50,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2024,7,30]],"date-time":"2024-07-30T00:00:00Z","timestamp":1722297600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,7,30]],"date-time":"2024-07-30T00:00:00Z","timestamp":1722297600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"the National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["62271160"],"award-info":[{"award-number":["62271160"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100005046","name":"the Natural Science Foundation of Heilongjiang Province of China","doi-asserted-by":"crossref","award":["LH2021F011"],"award-info":[{"award-number":["LH2021F011"]}],"id":[{"id":"10.13039\/501100005046","id-type":"DOI","asserted-by":"crossref"}]},{"name":"the Fundamental Research Funds for the Central Universities of China","award":["3072022TS0801"],"award-info":[{"award-number":["3072022TS0801"]}]},{"name":"the Key Laboratory Open Fund","award":["AMCIT2103-03"],"award-info":[{"award-number":["AMCIT2103-03"]}]},{"name":"the Natural Science Foundation of Jiangsu Province of China","award":["BK20210594"],"award-info":[{"award-number":["BK20210594"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2024,8]]},"DOI":"10.1007\/s00530-024-01418-5","type":"journal-article","created":{"date-parts":[[2024,7,30]],"date-time":"2024-07-30T05:11:03Z","timestamp":1722316263000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["GloFP-MSF: monocular scene flow estimation with global feature perception"],"prefix":"10.1007","volume":"30","author":[{"given":"Xuezhi","family":"Xiang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yu","family":"Cui","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xi","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mingliang","family":"Zhai","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Abdulmotaleb","family":"El Saddik","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,7,30]]},"reference":[{"key":"1418_CR1","doi-asserted-by":"crossref","unstructured":"Menze, M., Geiger, A.: Object scene flow for autonomous vehicles. In: Conference on Computer Vision and Pattern Recognition (CVPR) , 3061\u20133070 (2015)","DOI":"10.1109\/CVPR.2015.7298925"},{"key":"1418_CR2","doi-asserted-by":"crossref","unstructured":"Wang, P., Li, W., Gao, Z., Zhang, Y., Tang, C., Ogunbona, P.: Scene flow to action map: a new representation for RGB-D based action recognition with convolutional neural networks. In: Conference on Computer Vision and Pattern Recognition (CVPR), 595\u2013604 (2017)","DOI":"10.1109\/CVPR.2017.52"},{"key":"1418_CR3","doi-asserted-by":"crossref","unstructured":"Valgaerts, L., Chenglei, W., Bruhn, A., Seidel, H.-P., Theobalt, C: Lightweight binocular facial performance capture under uncontrolled lighting. ACM Trans. Graph., 1\u201311 (2012)","DOI":"10.1145\/2366145.2366206"},{"key":"1418_CR4","doi-asserted-by":"crossref","unstructured":"Hariat, M., Manzanera, A., Filliat, D.: Rebalancing gradient to improve self-supervised co-training of depth, odometry and optical flow predictions. In: 2022 IEEE\/CVF Workshop on Applications of Computer Vision(WACV), 1267\u20131276 (2022)","DOI":"10.1109\/WACV56688.2023.00132"},{"key":"1418_CR5","unstructured":"Mandal, D., Jain, A.: Unsupervised learning of depth, camera pose and optical flow from monocular video (2022). arXiv preprint arXiv:2205.09821"},{"key":"1418_CR6","doi-asserted-by":"crossref","unstructured":"Wang, G., Hu, Y., Liu, Z., Zhou, Y., Tomizuka, M., Zhan, W., Wang, H.: What matters for 3d scene flow network. In: European Conference on Computer Vision(ECCV), 38\u201355 (2022)","DOI":"10.1007\/978-3-031-19827-4_3"},{"key":"1418_CR7","doi-asserted-by":"crossref","unstructured":"Lang, I., Aiger, D., Cole, F., Avidan, S., Rubinstein, M.: SCOOP: self-supervised correspondence and optimization-based scene flow (2023). arXiv preprint arXiv:2211.14020","DOI":"10.1109\/CVPR52729.2023.00511"},{"key":"1418_CR8","doi-asserted-by":"crossref","unstructured":"Chen, Y., Schmid, C., Sminchisescu, C.: Self-supervised learning with eometric constraints in monocular video: Connecting flow, depth, and amera. In: IEEE\/CVF International Conference on Computer Vision(ICCV), 7063\u20137072 (2019)","DOI":"10.1109\/ICCV.2019.00716"},{"key":"1418_CR9","doi-asserted-by":"crossref","unstructured":"Luo, C., Yang, Z., Wang, P., Wang, Y., Xu, W., Ramakant N., Alan Y.: Every pixel counts++: joint learning of geometry and motion with 3D holistic understanding. In: IEEE Transactions on Pattern Analysis and Machine Intelligence (TPAMI), 2624\u20132641 (2020)","DOI":"10.1109\/TPAMI.2019.2930258"},{"key":"1418_CR10","doi-asserted-by":"crossref","unstructured":"Hur, J., Roth, S.: Self-supervised monocular scene flow estimation. In: Conference on Computer Vision and Pattern Recognition(CVPR), 7394\u20137403 (2020)","DOI":"10.1109\/CVPR42600.2020.00742"},{"key":"1418_CR11","doi-asserted-by":"crossref","unstructured":"Hur, J., Roth, S.: Self-supervised multi-frame monocular scene flow. In: IEEE\/CVF International Conference on Computer Vision(ICCV), 2684 (2021)","DOI":"10.1109\/CVPR46437.2021.00271"},{"key":"1418_CR12","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., Dehghani, M., Minderer, M., Heigold, G., Gelly, S., Uszkoreit, J., Houlsby, N.: An image is worth 16x16 words: transformers for image recognition at scale. In: International Conference on Learning Representations(ICLR) 5 (2021)"},{"key":"1418_CR13","unstructured":"Ali, A., Touvron, H., Caron, M., Bojanowski, P., Douze, M., Joulin, A., Laptev, I., Neverova, N., Synnaeve, G., Verbeek, J.: Xcit:Cross-covariance image transformers. In: Conference on Neural Information Processing Systems, 2\u20135 (2021)"},{"key":"1418_CR14","doi-asserted-by":"crossref","unstructured":"Bello, I., Zoph, B., Vaswani, A., Shlens, J.: Attention augmented convolutional networks. In: IEEE\/CVF International Conference on Computer Vision(ICCV), 3286\u20133295 (2019)","DOI":"10.1109\/ICCV.2019.00338"},{"key":"1418_CR15","doi-asserted-by":"crossref","unstructured":"Wu, H., Xiao, B., Codella, N., Liu, M., Dai, X., Yuan, L., Zhang, L.: Cvt: introducing convolutions to vision transformers. In: IEEE\/CVF International Conference on Computer Vision (ICCV), 2103.15808 (2021)","DOI":"10.1109\/ICCV48922.2021.00009"},{"key":"1418_CR16","doi-asserted-by":"crossref","unstructured":"Peng, Z., Huang, W., Gu, S., Xie, L., Wang, Y., Jiao, J., Ye, Q.: Conformer: Local features coupling global representations for visual recognition. In: IEEE\/CVF International Conference on Computer Vision(ICCV), 22\u201331 (2021)","DOI":"10.1109\/ICCV48922.2021.00042"},{"key":"1418_CR17","doi-asserted-by":"crossref","unstructured":"Vedula, S., Baker, S., Collins, R., Kanade, T., Rander, P.: Three-dimensional scene flow. In: IEEE conference on Computer Vision and Pattern Recognition(CVPR), 475\u2013486 (1999)","DOI":"10.1109\/ICCV.1999.790293"},{"key":"1418_CR18","doi-asserted-by":"crossref","unstructured":"Valgaerts, L., Bruhn, A., Zimmer, H., Weickert, J., Stoll, C., Theobalt, C.: Joint estimation of motion, structure and geometry from stereo sequences. In: Conference on European Conference on Computer Vision (ECCV) (2010)","DOI":"10.1007\/978-3-642-15561-1_41"},{"key":"1418_CR19","doi-asserted-by":"crossref","unstructured":"Basha, T., Moses, Y., Kiryati, N.: Multi-view scene flow estimation: a view centered variational approach. In: IEEE conference on Computer Vision and Pattern Recognition(CVPR) (2010)","DOI":"10.1109\/CVPR.2010.5539791"},{"key":"1418_CR20","doi-asserted-by":"crossref","unstructured":"Teed, Z., Deng, J.: RAFT-3D: Scene flow using rigid-motion embeddings. In: Conference on Computer Vision and Pattern Recognition (CVPR), 8375\u20138384 (2021)","DOI":"10.1109\/CVPR46437.2021.00827"},{"key":"1418_CR21","doi-asserted-by":"crossref","unstructured":"Mehl, L., Jahedi, A., Schmalfuss, J., Bruhn, A.: M-FUSE: multi-frame fusion for scene flow estimation. In: 2023 IEEE\/CVF Workshop on Applications of Computer Vision(WACV), 2022\u20132028 (2023)","DOI":"10.1109\/WACV56688.2023.00206"},{"key":"1418_CR22","doi-asserted-by":"crossref","unstructured":"Shen, Y., Hui, L., Xie, J., Yang, J.: Self-Supervised 3D scene flow estimation guided by superpoints. In: IEEE conference on Computer Vision and Pattern Recognition(CVPR), 5271\u20135280 (2023)","DOI":"10.1109\/CVPR52729.2023.00510"},{"key":"1418_CR23","unstructured":"Mandal, D., Jain, A.: Unsupervised learning of depth, camera pose and optical flow from monocular video (2022). arXiv preprint arXiv:2205.09821"},{"key":"1418_CR24","unstructured":"Vitor G., Kuan-Hui L.R., Ambrus, A.G.: Learning optical flow, depth and scene flow without real-world labels (2022). arXiv preprint arXiv:2203.15089"},{"key":"1418_CR25","doi-asserted-by":"crossref","unstructured":"Xiao, D., Yang, Q., Yang, B., Wei, W.: Monocular scene flow estimation via variational method. Multimed Tools Appl, 10575\u201310597 (2017)","DOI":"10.1007\/s11042-015-3091-6"},{"key":"1418_CR26","doi-asserted-by":"crossref","unstructured":"Brickwedde, F., Abraham, S., Mester, R.: Mono-SF: multi-view geometry meets single-view depth for monocular scene flow estimation of dynamic traffic scenes. In: IEEE conference on Computer Vision and Pattern Recognition(CVPR), 2780\u20132790 (2019)","DOI":"10.1109\/ICCV.2019.00287"},{"key":"1418_CR27","doi-asserted-by":"crossref","unstructured":"Sun, D., Yang, X., Liu, M.-Y., Kautz, J.: PWC-Net: CNNs for optical flow using pyramid, warping, and cost volume. In: IEEE conference on Computer Vision and Pattern Recognition(CVPR), 8934\u20138943 (2018)","DOI":"10.1109\/CVPR.2018.00931"},{"key":"1418_CR28","unstructured":"Shi, X., Chen, Z., Wang, H., Yeung, D.-Y., Wong, W.-K., Woo, W.-C.: Convolutional LSTM Network: a machine learning approach for precipitation nowcasting. In: Conference on Neural Information Processing Systems(NIPS), 802\u2013810 (2015)"},{"key":"1418_CR29","doi-asserted-by":"crossref","unstructured":"Bayramli, B., Hur, J., Lu, H.: Raft-msf: self-supervised monocular scene flow using recurrent optimizer. In: International Journal of Computer Vision(IJCV), 2757\u20132769 (2023)","DOI":"10.1007\/s11263-023-01828-4"},{"key":"1418_CR30","unstructured":"Devlin, J., Chang, M.-W., Lee, K., Toutanova, K.: BERT: pre-training of deep bidirectional transformers for language understanding. In: North American Chapter of the Association for Computational Linguistics: Human Language Technologies, 4171\u20134186 (2019)"},{"key":"1418_CR31","unstructured":"Liu, Y., Wu, Y.-H., Sun, G., Zhang, L., Chhatkuli, A., Van Gool, L.: Vision transformers with hierarchical attention (2021). arXiv preprint arXiv:2106.03180"},{"key":"1418_CR32","doi-asserted-by":"crossref","unstructured":"Jiang, S., Campbell, D., Lu, Y., Li, H., Hartley, R.: Learning to estimate hidden motions with global motion aggregation. In: IEEE\/CVF International Conference on Computer Vision(ICCV), 9772\u20139781 (2021)","DOI":"10.1109\/ICCV48922.2021.00963"},{"key":"1418_CR33","doi-asserted-by":"crossref","unstructured":"Xu, H., Zhang, J., Cai, J., Rezatofighi, H., Tao, D.: GMFlow: learning optical flow via global matching. In: IEEE\/CVF International Conference on Computer Vision(ICCV), 8121\u20138130 (2021)","DOI":"10.1109\/CVPR52688.2022.00795"},{"key":"1418_CR34","doi-asserted-by":"crossref","unstructured":"Teed, Z., Deng, J.: Raft: recurrent allpairs field transforms for optical flow. In: Conference on European Conference on Computer Vision (ECCV), 402\u2013419 (2020)","DOI":"10.1007\/978-3-030-58536-5_24"},{"key":"1418_CR35","doi-asserted-by":"crossref","unstructured":"Huang, Z., Shi, X., Zhang, C., Wang, Q.K.C.C., Hongwei Q., Jifeng D., Hongsheng L.: Flowformer: a transformer architecture for optical flow. In: Conference on European Conference on Computer Vision (ECCV), 13677\u201313675 (2022)","DOI":"10.1007\/978-3-031-19790-1_40"},{"key":"1418_CR36","doi-asserted-by":"crossref","unstructured":"Chen, Y., Zhu, D., Shi, W., Zhang, G., Zhang, T., Zhang, X., Li, J.: MFCFlow : a motion feature compensated multi-frame recurrent network for optical flow estimation. In: IEEE\/CVF Winter Conference on Applications of Computer Vision(WACV), 5068\u20135077 (2023)","DOI":"10.1109\/WACV56688.2023.00504"},{"key":"1418_CR37","doi-asserted-by":"crossref","unstructured":"Xu, H., Zhang, J., Cai, J., Rezatofighi, H., Yu, F., Tao, D., Geiger, A.: Unifying flow, stereo and depth estimation. In: IEEE Transactions on Pattern Analysis and Machine Intelligence (TPAMI), 13941\u201313958 (2023)","DOI":"10.1109\/TPAMI.2023.3298645"},{"key":"1418_CR38","doi-asserted-by":"crossref","unstructured":"Woo, S., Debnath, S., Ronghang, H., Chen, X., Liu, Z.: In So Kweon, Saining Xie: ConvNeXt V2: Co-designing and Scaling ConvNets with Masked Autoencoders. In: Conference on Computer Vision and Pattern Recognition (CVPR) 16133\u201316142 (2023)","DOI":"10.1109\/CVPR52729.2023.01548"},{"key":"1418_CR39","doi-asserted-by":"crossref","unstructured":"Srinivas, A., Lin, T.-Y., Parmar, N., Shlens, J., Abbeel, P., Vaswani, A.: Bottleneck transformers for visual recognition. In: Conference on Computer Vision and Pattern Recognition(CVPR), 16519\u201316529 (2021)","DOI":"10.1109\/CVPR46437.2021.01625"},{"key":"1418_CR40","unstructured":"Andreas, G., Philip, L., Christoph, S., Raquel, U.: Vision meets robotics: the kitti dataset. In: Conference on Neural Information Processing Systems(NIPS) (2013)"},{"key":"1418_CR41","unstructured":"Kingma, D.P., Ba, J.: Adam: a method for stochastic optimization. In: Conference on International Conference on Computer Vision(ICCV), 1842\u20131847 (2015)"},{"key":"1418_CR42","doi-asserted-by":"crossref","unstructured":"Zou, Y., Luo, Z., Huang, J.-B.: Df-net: unsupervised joint learning of depth and flow using cross-task consistency. In: Conference on European Conference on Computer Vision (ECCV), 36\u201353 (2019)","DOI":"10.1007\/978-3-030-01228-1_3"},{"key":"1418_CR43","doi-asserted-by":"crossref","unstructured":"Godard, Cl\u00e9ment, A., Oisin M., Brostow, G.J.: Unsupervised monocular depth estimation with left-right consistency. In: Conference on Computer Vision and Pattern Recognition(CVPR), 270\u2013279 (2018)","DOI":"10.1109\/CVPR.2017.699"},{"key":"1418_CR44","doi-asserted-by":"crossref","unstructured":"Yang, Z., Wang, P., Wang, Y., Xu, W., Nevatia, R.: Every pixel counts: unsupervised geometry learning with holistic 3d motion understanding. In: Conference on European Conference on Computer Vision (ECCV), 691\u2013709 (2018)","DOI":"10.1007\/978-3-030-11021-5_43"},{"key":"1418_CR45","doi-asserted-by":"crossref","unstructured":"Luo, C., Yang, Z., Wang, P., Wang, Y., Xu, W., Nevatia, R., Yuille, A.: Every pixel counts++: joint learning of geometry and motion with 3D holistic understanding. In: IEEE Transactions on Pattern Analysis and Machine Intelligence (TPAMI), 2624\u20132641 (2020)","DOI":"10.1109\/TPAMI.2019.2930258"},{"key":"1418_CR46","doi-asserted-by":"crossref","unstructured":"Cheng, Z., Liang, J., Choi, H., Tao, G., Cao, Z., Liu, D., Zhang, X.: Physical attack on monocular depth estimation with optimal adversarial patches. In: Conference on European Conference on Computer Vision (ECCV), 514\u2013532 (2022)","DOI":"10.1007\/978-3-031-19839-7_30"},{"key":"1418_CR47","doi-asserted-by":"crossref","unstructured":"Cheng, Z., Liang, J., Tao, G., Liu, D., Zhang, X.: Adversarial training of self-supervised monocular depth estimation against physical-world attacks. In: International Conference on Learning Representations (ICLR) (2023)","DOI":"10.1109\/TPAMI.2024.3412632"},{"key":"1418_CR48","unstructured":"Cheng, Z., Choi, H., Feng, S., Liang, J., Tao, G., Liu, D., Zuzak, M., Zhang, X.: Fusion is not enough: single-modal attacks to compromise fusion models in autonomous driving. In: International Conference on Learning Representations (ICLR) (2024)"},{"key":"1418_CR49","doi-asserted-by":"crossref","unstructured":"Liu, D., Cui, Y., Tan, W., Chen, Y.: SG-Net: Spatial granularity network for one-stage video instance segmentation. In: Conference on Computer Vision and Pattern Recognition(CVPR), 9816\u20139825 (2021)","DOI":"10.1109\/CVPR46437.2021.00969"},{"key":"1418_CR50","doi-asserted-by":"crossref","unstructured":"Cui, Y., Yan, L., Cao, Z., Liu, D.: TF-Blender: temporal feature blender for video object detection. In: Conference on International Conference on Computer Vision(ICCV), 8138\u20138147 (2021)","DOI":"10.1109\/ICCV48922.2021.00803"}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-024-01418-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-024-01418-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-024-01418-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,8,22]],"date-time":"2024-08-22T04:37:10Z","timestamp":1724301430000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-024-01418-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,7,30]]},"references-count":50,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2024,8]]}},"alternative-id":["1418"],"URL":"https:\/\/doi.org\/10.1007\/s00530-024-01418-5","relation":{"has-preprint":[{"id-type":"doi","id":"10.21203\/rs.3.rs-3856948\/v1","asserted-by":"object"}]},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"value":"0942-4962","type":"print"},{"value":"1432-1882","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,7,30]]},"assertion":[{"value":"12 January 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 July 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"30 July 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"227"}}