{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,10]],"date-time":"2026-07-10T16:52:33Z","timestamp":1783702353227,"version":"3.55.0"},"reference-count":64,"publisher":"Springer Science and Business Media LLC","issue":"10","license":[{"start":{"date-parts":[[2024,5,19]],"date-time":"2024-05-19T00:00:00Z","timestamp":1716076800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,5,19]],"date-time":"2024-05-19T00:00:00Z","timestamp":1716076800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62271414"],"award-info":[{"award-number":["62271414"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100019540","name":"Science Fund for Distinguished Young Scholars of Zhejiang Province","doi-asserted-by":"publisher","award":["LR23F010001"],"award-info":[{"award-number":["LR23F010001"]}],"id":[{"id":"10.13039\/501100019540","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2024,10]]},"DOI":"10.1007\/s11263-024-02101-y","type":"journal-article","created":{"date-parts":[[2024,5,19]],"date-time":"2024-05-19T15:01:12Z","timestamp":1716130872000},"page":"4521-4540","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":23,"title":["Hybrid CNN-Transformer Architecture for Efficient Large-Scale Video Snapshot Compressive Imaging"],"prefix":"10.1007","volume":"132","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2308-4388","authenticated-orcid":false,"given":"Miao","family":"Cao","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lishun","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mingyu","family":"Zhu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xin","family":"Yuan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,5,19]]},"reference":[{"key":"2101_CR1","unstructured":"Ba, J.L., Kiros, J.R., Hinton, G.E. (2016) Layer normalization. Advances in NIPS 2016 Deep Learning Symposium"},{"key":"2101_CR2","doi-asserted-by":"publisher","first-page":"8554","DOI":"10.1109\/TMM.2023.3238522","volume":"25","author":"Q Bao","year":"2023","unstructured":"Bao, Q., Liu, Y., Gang, B., et al. (2023). SCTANet: A spatial attention-guided CNN-transformer aggregation network for deep face image super-resolution. IEEE Transactions on Multimedia, 25, 8554\u20138565.","journal-title":"IEEE Transactions on Multimedia"},{"key":"2101_CR3","unstructured":"Behrmann, J., Grathwohl, W., Chen, R.T., et\u00a0al. (2019) Invertible residual networks. In: International Conference on Machine Learning, PMLR, pp. 573\u2013582."},{"key":"2101_CR4","unstructured":"Bertasius, G., Wang, H., Torresani, L. (2021) Is space-time attention all you need for video understanding? In: International Conference on Machine Learning, pp.\u00a04."},{"key":"2101_CR5","doi-asserted-by":"crossref","unstructured":"Cai, Y., Lin, J., Hu, X., et\u00a0al. (2022) Mask-guided spectral-wise transformer for efficient hyperspectral image reconstruction. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 17502\u201317511.","DOI":"10.1109\/CVPR52688.2022.01698"},{"issue":"2","key":"2101_CR6","doi-asserted-by":"publisher","first-page":"489","DOI":"10.1109\/TIT.2005.862083","volume":"52","author":"EJ Cand\u00e8s","year":"2006","unstructured":"Cand\u00e8s, E. J., Romberg, J., & Tao, T. (2006). Robust uncertainty principles: Exact signal reconstruction from highly incomplete frequency information. IEEE Transactions on Information Theory, 52(2), 489\u2013509.","journal-title":"IEEE Transactions on Information Theory"},{"key":"2101_CR7","first-page":"3154","volume":"45","author":"KC Chan","year":"2022","unstructured":"Chan, K. C., Xu, X., Wang, X., et al. (2022). Glean: Generative latent bank for image super-resolution and beyond. IEEE Transactions on Pattern Analysis and Machine Intelligence, 45, 3154\u20133168.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2101_CR8","doi-asserted-by":"crossref","unstructured":"Chang, Y.L., Liu. Z.Y., Lee, K.Y., et\u00a0al (2019) Free-form video inpainting with 3d gated convolution and temporal Patchgan. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 9066\u20139075.","DOI":"10.1109\/ICCV.2019.00916"},{"key":"2101_CR9","doi-asserted-by":"crossref","unstructured":"Chen, X., Pan, J., Lu, J., et\u00a0al (2023) Hybrid CNN-transformer feature fusion for single image Deraining. In: Proceedings of the AAAI Conference on Artificial Intelligence, pp. 378\u2013386.","DOI":"10.1609\/aaai.v37i1.25111"},{"key":"2101_CR10","doi-asserted-by":"crossref","unstructured":"Cheng, Z., Lu, R., Wang, Z., et\u00a0al. (2020) BIRNAT: Bidirectional recurrent neural networks with adversarial training for video snapshot compressive imaging. In: European Conference on Computer Vision. Springer, pp. 258\u2013275.","DOI":"10.1007\/978-3-030-58586-0_16"},{"key":"2101_CR11","doi-asserted-by":"crossref","unstructured":"Cheng, Z., Chen, B., Liu, G., et\u00a0al (2021) Memory-efficient network for large-scale video compressive sensing. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16246\u201316255.","DOI":"10.1109\/CVPR46437.2021.01598"},{"key":"2101_CR12","unstructured":"Chu, X., Tian, Z., Zhang, B., et\u00a0al (2022) Conditional positional encodings for vision transformers. In: International Conference on Learning Representations"},{"key":"2101_CR13","doi-asserted-by":"crossref","unstructured":"Dong, X., Bao, J., & Chen, D., et\u00a0al (2022) Cswin transformer: A general vision transformer backbone with cross-shaped windows. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12124\u201312134.","DOI":"10.1109\/CVPR52688.2022.01181"},{"issue":"4","key":"2101_CR14","doi-asserted-by":"publisher","first-page":"1289","DOI":"10.1109\/TIT.2006.871582","volume":"52","author":"DL Donoho","year":"2006","unstructured":"Donoho, D. L. (2006). Compressed sensing. IEEE Transactions on Information Theory, 52(4), 1289\u20131306.","journal-title":"IEEE Transactions on Information Theory"},{"key":"2101_CR15","unstructured":"Dosovitskiy, A., Beyer, L., Kolesnikov, A., et\u00a0al (2020) An image is worth 16x16 words: Transformers for image recognition at scale. In: International Conference on Learning Representations"},{"key":"2101_CR16","doi-asserted-by":"publisher","first-page":"1978","DOI":"10.1109\/TIP.2023.3261747","volume":"32","author":"G Gao","year":"2023","unstructured":"Gao, G., Xu, Z., Li, J., et al. (2023). CTCNet: A CNN-transformer cooperation network for face image super-resolution. IEEE Transactions on Image Processing, 32, 1978\u20131991.","journal-title":"IEEE Transactions on Image Processing"},{"issue":"2","key":"2101_CR17","doi-asserted-by":"publisher","first-page":"652","DOI":"10.1109\/TPAMI.2019.2938758","volume":"43","author":"SH Gao","year":"2019","unstructured":"Gao, S. H., Cheng, M. M., Zhao, K., et al. (2019). Res2net: A new multi-scale backbone architecture. IEEE Transactions on Pattern Analysis and Machine Intelligence, 43(2), 652\u2013662.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2101_CR18","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., et\u00a0al (2016) Deep residual learning for image recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778.","DOI":"10.1109\/CVPR.2016.90"},{"key":"2101_CR19","unstructured":"Hendrycks, D., Gimpel, K. (2016) Gaussian error linear units (gelus). arXiv preprint arXiv:1606.08415"},{"key":"2101_CR20","doi-asserted-by":"crossref","unstructured":"Hitomi, Y., Gu, J., Gupta, M., et\u00a0al (2011) Video from a single coded exposure photograph using a learned over-complete dictionary. In: 2011 International Conference on Computer Vision. IEEE, pp. 287\u2013294.","DOI":"10.1109\/ICCV.2011.6126254"},{"key":"2101_CR21","doi-asserted-by":"crossref","unstructured":"Huang, G., Liu, Z., & Van Der Maaten, L., et\u00a0al (2017) Densely connected convolutional networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4700\u20134708.","DOI":"10.1109\/CVPR.2017.243"},{"key":"2101_CR22","unstructured":"Ioffe, S., & Szegedy, C. (2015) Batch normalization: Accelerating deep network training by reducing internal covariate shift. In: International Conference on Machine Learning, PMLR, pp. 448\u2013456."},{"key":"2101_CR23","unstructured":"Islam, M. A., Jia, S., & Bruce, N. D. (2020). How much position information do convolutional neural networks encode?"},{"issue":"1","key":"2101_CR24","doi-asserted-by":"publisher","first-page":"221","DOI":"10.1109\/TPAMI.2012.59","volume":"35","author":"S Ji","year":"2012","unstructured":"Ji, S., Xu, W., Yang, M., et al. (2012). 3d convolutional neural networks for human action recognition. IEEE Transactions on Pattern Analysis and Machine Intelligence, 35(1), 221\u2013231.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2101_CR25","unstructured":"Kingma, D.P., & Ba, J. (2015) Adam: A method for stochastic optimization. In: International Conference on Learning Representations"},{"issue":"10","key":"2101_CR26","doi-asserted-by":"publisher","first-page":"2385","DOI":"10.1007\/s11263-022-01651-3","volume":"130","author":"G Kordopatis-Zilos","year":"2022","unstructured":"Kordopatis-Zilos, G., Tzelepis, C., Papadopoulos, S., et al. (2022). Dns: Distill-and-select for efficient and accurate video indexing and retrieval. International Journal of Computer Vision, 130(10), 2385\u20132407.","journal-title":"International Journal of Computer Vision"},{"key":"2101_CR27","first-page":"1097","volume":"25","author":"A Krizhevsky","year":"2012","unstructured":"Krizhevsky, A., Sutskever, I., & Hinton, G. E. (2012). Imagenet classification with deep convolutional neural networks. Advances in Neural Information Processing Systems, 25, 1097\u20131105.","journal-title":"Advances in Neural Information Processing Systems"},{"issue":"12","key":"2101_CR28","doi-asserted-by":"publisher","first-page":"9396","DOI":"10.1109\/TPAMI.2021.3126387","volume":"44","author":"C Li","year":"2021","unstructured":"Li, C., Guo, C., Han, L., et al. (2021). Low-light image and video enhancement using deep learning: A survey. IEEE Transactions on Pattern Analysis and Machine Intelligence, 44(12), 9396\u20139416.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2101_CR29","doi-asserted-by":"crossref","unstructured":"Liu, C., Kim, K., & Gu, J., et\u00a0al (2019) Planercnn: 3d plane detection and reconstruction from a single image. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 4450\u20134459.","DOI":"10.1109\/CVPR.2019.00458"},{"issue":"2","key":"2101_CR30","first-page":"248","volume":"36","author":"D Liu","year":"2013","unstructured":"Liu, D., Gu, J., Hitomi, Y., et al. (2013). Efficient space-time sampling with pixel-wise coded exposure for high-speed imaging. IEEE Transactions on Pattern Analysis and Machine Intelligence, 36(2), 248\u2013260.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"issue":"12","key":"2101_CR31","doi-asserted-by":"publisher","first-page":"2990","DOI":"10.1109\/TPAMI.2018.2873587","volume":"41","author":"Y Liu","year":"2018","unstructured":"Liu, Y., Yuan, X., Suo, J., et al. (2018). Rank minimization for snapshot compressive imaging. IEEE Transactions on Pattern Analysis and Machine Intelligence, 41(12), 2990\u20133006.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2101_CR32","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lin, Y., & Cao, Y., et\u00a0al (2021) Swin transformer: Hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 10012\u201310022.","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"2101_CR33","doi-asserted-by":"crossref","unstructured":"Liu, Z., Ning, J., & Cao, Y., et\u00a0al (2022) Video swin transformer. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3202\u20133211.","DOI":"10.1109\/CVPR52688.2022.00320"},{"issue":"9","key":"2101_CR34","doi-asserted-by":"publisher","first-page":"10526","DOI":"10.1364\/OE.21.010526","volume":"21","author":"P Llull","year":"2013","unstructured":"Llull, P., Liao, X., Yuan, X., et al. (2013). Coded aperture compressive temporal imaging. Optics Express, 21(9), 10526\u201310545.","journal-title":"Optics Express"},{"key":"2101_CR35","unstructured":"Maas, A.L., Hannun, A.Y., Ng, A.Y., et\u00a0al (2013) Rectifier nonlinearities improve neural network acoustic models. In: International Conference on Machine Learning, Citeseer, pp. 3."},{"key":"2101_CR36","unstructured":"Micikevicius, P., Narang, S., Alben, J., et\u00a0al (2017) Mixed precision training. In: International Conference on Learning Representations"},{"issue":"4","key":"2101_CR37","doi-asserted-by":"publisher","first-page":"783","DOI":"10.1007\/s11263-019-01283-0","volume":"128","author":"J Park","year":"2020","unstructured":"Park, J., Woo, S., Lee, J. Y., et al. (2020). A simple and light-weight attention module for convolutional neural networks. International Journal of Computer Vision, 128(4), 783\u2013798.","journal-title":"International Journal of Computer Vision"},{"key":"2101_CR38","unstructured":"Pont-Tuset, J., Perazzi, F., Caelles, S., et\u00a0al (2017) The 2017 Davis challenge on video object segmentation. arXiv preprint arXiv:1704.00675"},{"issue":"3","key":"2101_CR39","doi-asserted-by":"publisher","first-page":"30801","DOI":"10.1063\/1.5140721","volume":"5","author":"M Qiao","year":"2020","unstructured":"Qiao, M., Meng, Z., Ma, J., et al. (2020). Deep learning for video compressive sensing. APL Photonics, 5(3), 30801.","journal-title":"APL Photonics"},{"key":"2101_CR40","doi-asserted-by":"crossref","unstructured":"Shi, W., Caballero, J., Husz\u00e1r, F., et\u00a0al (2016) Real-time single image and video super-resolution using an efficient sub-pixel convolutional neural network. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1874\u20131883.","DOI":"10.1109\/CVPR.2016.207"},{"key":"2101_CR41","doi-asserted-by":"crossref","unstructured":"Wang, C.Y., Liao, H.Y.M., Wu, Y.H., et\u00a0al (2020) CSPNet: A new backbone that can enhance learning capability of CNN. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition Workshops, pp. 390\u2013391.","DOI":"10.1109\/CVPRW50498.2020.00203"},{"key":"2101_CR42","first-page":"9072","volume":"45","author":"L Wang","year":"2022","unstructured":"Wang, L., Cao, M., Zhong, Y., et al. (2022). Spatial-temporal transformer for video snapshot compressive imaging. IEEE Transactions on Pattern Analysis and Machine Intelligence, 45, 9072\u20139089.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"issue":"8","key":"2101_CR43","doi-asserted-by":"publisher","first-page":"1848","DOI":"10.1364\/PRJ.458231","volume":"10","author":"L Wang","year":"2022","unstructured":"Wang, L., Wu, Z., Zhong, Y., et al. (2022). Snapshot spectral compressive imaging reconstruction using convolution and contextual transformer. Photonics Research, 10(8), 1848\u20131858.","journal-title":"Photonics Research"},{"key":"2101_CR44","doi-asserted-by":"crossref","unstructured":"Wang, L., Cao, M., Yuan, X. (2023) Efficientsci: Densely connected network with space-time factorization for large-scale video snapshot compressive imaging. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 18477\u201318486.","DOI":"10.1109\/CVPR52729.2023.01772"},{"issue":"4","key":"2101_CR45","doi-asserted-by":"publisher","first-page":"600","DOI":"10.1109\/TIP.2003.819861","volume":"13","author":"Z Wang","year":"2004","unstructured":"Wang, Z., Bovik, A. C., Sheikh, H. R., et al. (2004). Image quality assessment: From error visibility to structural similarity. IEEE Transactions on Image Processing, 13(4), 600\u2013612.","journal-title":"IEEE Transactions on Image Processing"},{"key":"2101_CR46","doi-asserted-by":"crossref","unstructured":"Wang, Z., Zhang, H., Cheng, Z., et\u00a0al (2021) MetaSCI: Scalable and adaptive reconstruction for video compressive sensing. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2083\u20132092.","DOI":"10.1109\/CVPR46437.2021.00212"},{"key":"2101_CR47","unstructured":"Wu, Z., Zhang, J., & Mou, C. (2021) Dense deep unfolding network with 3D-CNN prior for snapshot compressive imaging. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 4892\u20134901."},{"key":"2101_CR48","doi-asserted-by":"publisher","first-page":"1662","DOI":"10.1007\/s11263-023-01777-y","volume":"131","author":"Z Wu","year":"2023","unstructured":"Wu, Z., Yang, C., Su, X., et al. (2023). Adaptive deep pnp algorithm for video snapshot compressive imaging. International Journal of Computer Vision, 131, 1662\u20131679.","journal-title":"International Journal of Computer Vision"},{"key":"2101_CR49","doi-asserted-by":"crossref","unstructured":"Xie, S., Girshick, R., Doll\u00e1r, P., et\u00a0al (2017) Aggregated residual transformations for deep neural networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1492\u20131500.","DOI":"10.1109\/CVPR.2017.634"},{"key":"2101_CR50","doi-asserted-by":"crossref","unstructured":"Yang, C., Zhang, S., Yuan, X. (2022) Ensemble learning priors driven deep unfolding for scalable video snapshot compressive imaging. In: European Conference on Computer Vision","DOI":"10.1007\/978-3-031-20050-2_35"},{"issue":"1","key":"2101_CR51","doi-asserted-by":"publisher","first-page":"106","DOI":"10.1109\/TIP.2014.2365720","volume":"24","author":"J Yang","year":"2014","unstructured":"Yang, J., Liao, X., Yuan, X., et al. (2014). Compressive sensing by learning a Gaussian mixture model from measurements. IEEE Transactions on Image Processing, 24(1), 106\u2013119.","journal-title":"IEEE Transactions on Image Processing"},{"key":"2101_CR52","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2021.107899","volume":"115","author":"SK Yeom","year":"2021","unstructured":"Yeom, S. K., Seegerer, P., Lapuschkin, S., et al. (2021). Pruning by explaining: A novel criterion for deep neural network pruning. Pattern Recognition, 115, 107899.","journal-title":"Pattern Recognition"},{"issue":"6","key":"2101_CR53","doi-asserted-by":"publisher","first-page":"1307","DOI":"10.1007\/s11263-023-01758-1","volume":"131","author":"Z Yu","year":"2023","unstructured":"Yu, Z., Shen, Y., Shi, J., et al. (2023). Physformer++: Facial video-based physiological measurement with slowfast temporal difference transformer. International Journal of Computer Vision, 131(6), 1307\u20131330.","journal-title":"International Journal of Computer Vision"},{"key":"2101_CR54","doi-asserted-by":"crossref","unstructured":"Yuan, X. (2016) Generalized alternating projection based total variation minimization for compressive sensing. In: IEEE International Conference on Image Processing. IEEE, pp. 2539\u20132543.","DOI":"10.1109\/ICIP.2016.7532817"},{"issue":"6","key":"2101_CR55","doi-asserted-by":"publisher","first-page":"964","DOI":"10.1109\/JSTSP.2015.2411575","volume":"9","author":"X Yuan","year":"2015","unstructured":"Yuan, X., Tsai, T. H., Zhu, R., et al. (2015). Compressive hyperspectral imaging with side information. IEEE Journal of Selected Topics in Signal Processing, 9(6), 964\u2013976.","journal-title":"IEEE Journal of Selected Topics in Signal Processing"},{"key":"2101_CR56","doi-asserted-by":"crossref","unstructured":"Yuan, X., Liu, Y., Suo, J., et\u00a0al (2020) Plug-and-play algorithms for large-scale snapshot compressive imaging. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1447\u20131457.","DOI":"10.1109\/CVPR42600.2020.00152"},{"issue":"2","key":"2101_CR57","doi-asserted-by":"publisher","first-page":"65","DOI":"10.1109\/MSP.2020.3023869","volume":"38","author":"X Yuan","year":"2021","unstructured":"Yuan, X., Brady, D. J., & Katsaggelos, A. K. (2021). Snapshot compressive imaging: Theory, algorithms, and applications. IEEE Signal Processing Magazine, 38(2), 65\u201388.","journal-title":"IEEE Signal Processing Magazine"},{"issue":"01","key":"2101_CR58","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1109\/TPAMI.2021.3068277","volume":"44","author":"X Yuan","year":"2021","unstructured":"Yuan, X., Liu, Y., Suo, J., et al. (2021). Plug-and-play algorithms for video snapshot compressive imaging. IEEE Transactions on Pattern Analysis and Machine Intelligence, 44(01), 1\u20131.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"2101_CR59","doi-asserted-by":"crossref","unstructured":"Zamir, S.W., Arora, A., Khan, S., et\u00a0al (2022) Restormer: Efficient transformer for high-resolution image restoration. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5728\u20135739.","DOI":"10.1109\/CVPR52688.2022.00564"},{"issue":"5","key":"2101_CR60","doi-asserted-by":"publisher","first-page":"1141","DOI":"10.1007\/s11263-022-01739-w","volume":"131","author":"Q Zhang","year":"2023","unstructured":"Zhang, Q., Xu, Y., Zhang, J., et al. (2023). Vitaev2: Vision transformer advanced by exploring inductive bias for image recognition and beyond. International Journal of Computer Vision, 131(5), 1141\u20131162.","journal-title":"International Journal of Computer Vision"},{"key":"2101_CR61","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Jiang, Y., Jiang, J., et\u00a0al (2021a) Star: A structure-aware lightweight transformer for real-time image enhancement. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 4106\u20134115.","DOI":"10.1109\/ICCV48922.2021.00407"},{"key":"2101_CR62","unstructured":"Zhang, Z., Shao, W., Gu, J., et\u00a0al (2021b) Differentiable dynamic quantization with mixed precision and adaptive resolution. In: International Conference on Machine Learning, PMLR, pp. 12546\u201312556."},{"key":"2101_CR63","doi-asserted-by":"crossref","unstructured":"Zhang, Z., Jiang, Y., Shao, W., et\u00a0al (2023b) Real-time controllable denoising for image and video. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14028\u201314038.","DOI":"10.1109\/CVPR52729.2023.01348"},{"issue":"9","key":"2101_CR64","doi-asserted-by":"publisher","first-page":"2081","DOI":"10.1007\/s11263-022-01638-0","volume":"130","author":"B Zhuang","year":"2022","unstructured":"Zhuang, B., Shen, C., Tan, M., et al. (2022). Structured binary neural networks for image recognition. International Journal of Computer Vision, 130(9), 2081\u20132102.","journal-title":"International Journal of Computer Vision"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-024-02101-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-024-02101-y\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-024-02101-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,4]],"date-time":"2024-10-04T06:23:23Z","timestamp":1728023003000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-024-02101-y"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,5,19]]},"references-count":64,"journal-issue":{"issue":"10","published-print":{"date-parts":[[2024,10]]}},"alternative-id":["2101"],"URL":"https:\/\/doi.org\/10.1007\/s11263-024-02101-y","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,5,19]]},"assertion":[{"value":"31 May 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"23 April 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"19 May 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no Conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}