{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,23]],"date-time":"2026-07-23T01:43:30Z","timestamp":1784771010455,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":69,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,10,10]],"date-time":"2022-10-10T00:00:00Z","timestamp":1665360000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"the General Program of China Postdoctoral Science Foundation","award":["Grant No. 2020M683490"],"award-info":[{"award-number":["Grant No. 2020M683490"]}]},{"name":"the National Key Research and Development Program of China","award":["Grant No. 2017YFA0700800"],"award-info":[{"award-number":["Grant No. 2017YFA0700800"]}]},{"name":"the Youth program of Shaanxi Natural Science Foundation","award":["No. 2021JQ-054"],"award-info":[{"award-number":["No. 2021JQ-054"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,10,10]]},"DOI":"10.1145\/3503161.3548446","type":"proceedings-article","created":{"date-parts":[[2022,10,10]],"date-time":"2022-10-10T15:42:35Z","timestamp":1665416555000},"page":"6559-6568","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":54,"title":["T-former: An Efficient Transformer for Image Inpainting"],"prefix":"10.1145","author":[{"given":"Ye","family":"Deng","sequence":"first","affiliation":[{"name":"Xi'an Jiaotong University, Xi'an, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Siqi","family":"Hui","sequence":"additional","affiliation":[{"name":"Xi'an Jiaotong University, Xi'an, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sanping","family":"Zhou","sequence":"additional","affiliation":[{"name":"Xi'an Jiaotong University &amp; Shunan Academy of Artificial Intelligence, Xi'an and Ningbo, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Deyu","family":"Meng","sequence":"additional","affiliation":[{"name":"Xi'an Jiaotong University, Xi'an, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jinjun","family":"Wang","sequence":"additional","affiliation":[{"name":"Xi'an Jiaotong University, Xi'an, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2022,10,10]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/83.935036"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/1531326.1531330"},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2006.877067"},{"key":"e_1_3_2_2_4_1","volume-title":"Image Inpainting. In Proceedings of the 27th Annual Conference on Computer Graphics and Interactive Techniques (SIGGRAPH '00)","author":"Bertalmio Marcelo","year":"2000","unstructured":"Marcelo Bertalmio , Guillermo Sapiro , Vincent Caselles , and Coloma Ballester . 2000 . Image Inpainting. In Proceedings of the 27th Annual Conference on Computer Graphics and Interactive Techniques (SIGGRAPH '00) . ACM Press\/Addison-Wesley Publishing Co., USA, 417--424. https:\/\/doi.org\/10.1145\/344779.344972 Marcelo Bertalmio, Guillermo Sapiro, Vincent Caselles, and Coloma Ballester. 2000. Image Inpainting. In Proceedings of the 27th Annual Conference on Computer Graphics and Interactive Techniques (SIGGRAPH '00). ACM Press\/Addison-Wesley Publishing Co., USA, 417--424. https:\/\/doi.org\/10.1145\/344779.344972"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2015.2411437"},{"key":"e_1_3_2_2_6_1","volume-title":"European Conference on Computer Vision, Andrea Vedaldi, Horst Bischof, Thomas Brox, and Jan-Michael Frahm (Eds.)","author":"Carion Nicolas","unstructured":"Nicolas Carion , Francisco Massa , Gabriel Synnaeve , Nicolas Usunier , Alexander Kirillov , and Sergey Zagoruyko . 2020. End-to-End Object Detection with Transformers . In European Conference on Computer Vision, Andrea Vedaldi, Horst Bischof, Thomas Brox, and Jan-Michael Frahm (Eds.) . Springer International Publishing , 213--229. Nicolas Carion, Francisco Massa, Gabriel Synnaeve, Nicolas Usunier, Alexander Kirillov, and Sergey Zagoruyko. 2020. End-to-End Object Detection with Transformers. In European Conference on Computer Vision, Andrea Vedaldi, Horst Bischof, Thomas Brox, and Jan-Michael Frahm (Eds.). Springer International Publishing, 213--229."},{"key":"e_1_3_2_2_7_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 11315--11325","author":"Chang Huiwen","unstructured":"Huiwen Chang , Han Zhang , Lu Jiang , Ce Liu , and William T. Freeman . 2022. MaskGIT: Masked Generative Image Transformer . In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 11315--11325 . Huiwen Chang, Han Zhang, Lu Jiang, Ce Liu, and William T. Freeman. 2022. MaskGIT: Masked Generative Image Transformer. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 11315--11325."},{"key":"e_1_3_2_2_8_1","volume-title":"Dzmitry Bahdanau, and Yoshua Bengio.","author":"Cho Kyunghyun","year":"2014","unstructured":"Kyunghyun Cho , Bart Van Merri\u00ebnboer , Dzmitry Bahdanau, and Yoshua Bengio. 2014 . On the properties of neural machine translation: Encoder-decoder approaches. arXiv preprint arXiv:1409.1259 (2014). Kyunghyun Cho, Bart Van Merri\u00ebnboer, Dzmitry Bahdanau, and Yoshua Bengio. 2014. On the properties of neural machine translation: Encoder-decoder approaches. arXiv preprint arXiv:1409.1259 (2014)."},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2004.833105"},{"key":"e_1_3_2_2_10_1","volume-title":"Proceedings of the 34th International Conference on Machine Learning (Proceedings of Machine Learning Research","volume":"941","author":"Dauphin Yann N.","year":"2017","unstructured":"Yann N. Dauphin , Angela Fan , Michael Auli , and David Grangier . 2017 . Language Modeling with Gated Convolutional Networks . In Proceedings of the 34th International Conference on Machine Learning (Proceedings of Machine Learning Research , Vol. 70), Doina Precup and Yee Whye Teh (Eds.). PMLR, 933-- 941 . Yann N. Dauphin, Angela Fan, Michael Auli, and David Grangier. 2017. Language Modeling with Gated Convolutional Networks. In Proceedings of the 34th International Conference on Machine Learning (Proceedings of Machine Learning Research, Vol. 70), Doina Precup and Yee Whye Teh (Eds.). PMLR, 933--941."},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475426"},{"key":"e_1_3_2_2_13_1","volume-title":"International Conference on Learning Representations.","author":"Dosovitskiy Alexey","year":"2021","unstructured":"Alexey Dosovitskiy , Lucas Beyer , Alexander Kolesnikov , Dirk Weissenborn , Xiaohua Zhai , Thomas Unterthiner , Mostafa Dehghani , Matthias Minderer , Georg Heigold , Sylvain Gelly , Jakob Uszkoreit , and Neil Houlsby . 2021 . An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale . In International Conference on Learning Representations. Alexey Dosovitskiy, Lucas Beyer, Alexander Kolesnikov, Dirk Weissenborn, Xiaohua Zhai, Thomas Unterthiner, Mostafa Dehghani, Matthias Minderer, Georg Heigold, Sylvain Gelly, Jakob Uszkoreit, and Neil Houlsby. 2021. An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. In International Conference on Learning Representations."},{"key":"e_1_3_2_2_14_1","first-page":"3518","article-title":"ImageBART: Bidirectional Context with Multinomial Diffusion for Autoregressive Image Synthesis","volume":"34","author":"Esser Patrick","year":"2021","unstructured":"Patrick Esser , Robin Rombach , Andreas Blattmann , and Bjorn Ommer . 2021 b. ImageBART: Bidirectional Context with Multinomial Diffusion for Autoregressive Image Synthesis . In Advances in Neural Information Processing Systems , Vol. 34. 3518 -- 3532 . Patrick Esser, Robin Rombach, Andreas Blattmann, and Bjorn Ommer. 2021b. ImageBART: Bidirectional Context with Multinomial Diffusion for Autoregressive Image Synthesis. In Advances in Neural Information Processing Systems, Vol. 34. 3518--3532.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01268"},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.265"},{"key":"e_1_3_2_2_17_1","volume-title":"Weinberger (Eds.)","volume":"27","author":"Goodfellow Ian","year":"2014","unstructured":"Ian Goodfellow , Jean Pouget-Abadie , Mehdi Mirza , Bing Xu , David Warde-Farley , Sherjil Ozair , Aaron Courville , and Yoshua Bengio . 2014 . Generative Adversarial Nets. In Advances in Neural Information Processing Systems, Z. Ghahramani, M. Welling, C. Cortes, N. Lawrence, and K. Q . Weinberger (Eds.) , Vol. 27 . Curran Associates, Inc. Ian Goodfellow, Jean Pouget-Abadie, Mehdi Mirza, Bing Xu, David Warde-Farley, Sherjil Ozair, Aaron Courville, and Yoshua Bengio. 2014. Generative Adversarial Nets. In Advances in Neural Information Processing Systems, Z. Ghahramani, M. Welling, C. Cortes, N. Lawrence, and K. Q. Weinberger (Eds.), Vol. 27. Curran Associates, Inc."},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.01387"},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_2_20_1","volume-title":"Gaussian error linear units (gelus). arXiv preprint arXiv:1606.08415","author":"Hendrycks Dan","year":"2016","unstructured":"Dan Hendrycks and Kevin Gimpel . 2016. Gaussian error linear units (gelus). arXiv preprint arXiv:1606.08415 ( 2016 ). Dan Hendrycks and Kevin Gimpel. 2016. Gaussian error linear units (gelus). arXiv preprint arXiv:1606.08415 (2016)."},{"key":"e_1_3_2_2_21_1","volume-title":"Advances in Neural Information Processing Systems","volume":"30","author":"Heusel Martin","year":"2017","unstructured":"Martin Heusel , Hubert Ramsauer , Thomas Unterthiner , Bernhard Nessler , and Sepp Hochreiter . 2017 . GANs Trained by a Two Time-Scale Update Rule Converge to a Local Nash Equilibrium . In Advances in Neural Information Processing Systems , Vol. 30 . Martin Heusel, Hubert Ramsauer, Thomas Unterthiner, Bernhard Nessler, and Sepp Hochreiter. 2017. GANs Trained by a Two Time-Scale Update Rule Converge to a Local Nash Equilibrium. In Advances in Neural Information Processing Systems, Vol. 30."},{"key":"e_1_3_2_2_22_1","volume-title":"Long short-term memory. Neural computation","author":"Hochreiter Sepp","year":"1997","unstructured":"Sepp Hochreiter and J\u00fcrgen Schmidhuber . 1997. Long short-term memory. Neural computation , Vol. 9 , 8 ( 1997 ), 1735--1780. Sepp Hochreiter and J\u00fcrgen Schmidhuber. 1997. Long short-term memory. Neural computation, Vol. 9, 8 (1997), 1735--1780."},{"key":"e_1_3_2_2_23_1","volume-title":"Transformer Quality in Linear Time. arXiv preprint arXiv:2202.10447","author":"Hua Weizhe","year":"2022","unstructured":"Weizhe Hua , Zihang Dai , Hanxiao Liu , and Quoc V Le. 2022. Transformer Quality in Linear Time. arXiv preprint arXiv:2202.10447 ( 2022 ). Weizhe Hua, Zihang Dai, Hanxiao Liu, and Quoc V Le. 2022. Transformer Quality in Linear Time. arXiv preprint arXiv:2202.10447 (2022)."},{"key":"e_1_3_2_2_24_1","volume-title":"Globally and Locally Consistent Image Completion. ACM Transactions on Graphics (Proc. of SIGGRAPH 2017","volume":"36","author":"Iizuka Satoshi","year":"2017","unstructured":"Satoshi Iizuka , Edgar Simo-Serra , and Hiroshi Ishikawa . 2017 . Globally and Locally Consistent Image Completion. ACM Transactions on Graphics (Proc. of SIGGRAPH 2017 ), Vol. 36 , 4, Article 107 (2017), 107:1--107:14 pages. Satoshi Iizuka, Edgar Simo-Serra, and Hiroshi Ishikawa. 2017. Globally and Locally Consistent Image Completion. ACM Transactions on Graphics (Proc. of SIGGRAPH 2017), Vol. 36, 4, Article 107 (2017), 107:1--107:14 pages."},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46475-6_43"},{"key":"e_1_3_2_2_26_1","volume-title":"International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=Hk99zCeAb","author":"Karras Tero","year":"2018","unstructured":"Tero Karras , Timo Aila , Samuli Laine , and Jaakko Lehtinen . 2018 . Progressive Growing of GANs for Improved Quality, Stability, and Variation . In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=Hk99zCeAb Tero Karras, Timo Aila, Samuli Laine, and Jaakko Lehtinen. 2018. Progressive Growing of GANs for Improved Quality, Stability, and Variation. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=Hk99zCeAb"},{"key":"e_1_3_2_2_27_1","volume-title":"Proceedings of the 37th International Conference on Machine Learning (Proceedings of Machine Learning Research","author":"Katharopoulos Angelos","year":"2020","unstructured":"Angelos Katharopoulos , Apoorv Vyas , Nikolaos Pappas , and Francc ois Fleuret . 2020 . Transformers are RNNs: Fast Autoregressive Transformers with Linear Attention . In Proceedings of the 37th International Conference on Machine Learning (Proceedings of Machine Learning Research , Vol. 119), Hal Daum\u00e9 III and Aarti Singh (Eds.). PMLR, 5156--5165. Angelos Katharopoulos, Apoorv Vyas, Nikolaos Pappas, and Francc ois Fleuret. 2020. Transformers are RNNs: Fast Autoregressive Transformers with Linear Attention. In Proceedings of the 37th International Conference on Machine Learning (Proceedings of Machine Learning Research, Vol. 119), Hal Daum\u00e9 III and Aarti Singh (Eds.). PMLR, 5156--5165."},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2007.906269"},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00606"},{"key":"e_1_3_2_2_30_1","volume-title":"Recurrent Feature Reasoning for Image Inpainting. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR).","author":"Li Jingyuan","year":"2020","unstructured":"Jingyuan Li , Ning Wang , Lefei Zhang , Bo Du , and Dacheng Tao . 2020 . Recurrent Feature Reasoning for Image Inpainting. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). Jingyuan Li, Ning Wang, Lefei Zhang, Bo Du, and Dacheng Tao. 2020. Recurrent Feature Reasoning for Image Inpainting. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)."},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW54120.2021.00210"},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58583-9_41"},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00647"},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01252-6_6"},{"key":"e_1_3_2_2_35_1","volume-title":"European Conference on Computer Vision. Springer International Publishing, Cham, 725--741","author":"Liu Hongyu","year":"2020","unstructured":"Hongyu Liu , Bin Jiang , Yibing Song , Wei Huang , and Chao Yang . 2020 . Rethinking Image Inpainting via a Mutual Encoder-Decoder with Feature Equalizations . In European Conference on Computer Vision. Springer International Publishing, Cham, 725--741 . Hongyu Liu, Bin Jiang, Yibing Song, Wei Huang, and Chao Yang. 2020. Rethinking Image Inpainting via a Mutual Encoder-Decoder with Feature Equalizations. In European Conference on Computer Vision. Springer International Publishing, Cham, 725--741."},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00427"},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"e_1_3_2_2_38_1","volume-title":"Decoupled Weight Decay Regularization. In International Conference on Learning Representations.","author":"Loshchilov Ilya","year":"2019","unstructured":"Ilya Loshchilov and Frank Hutter . 2019 . Decoupled Weight Decay Regularization. In International Conference on Learning Representations. Ilya Loshchilov and Frank Hutter. 2019. Decoupled Weight Decay Regularization. In International Conference on Learning Representations."},{"key":"e_1_3_2_2_39_1","volume-title":"Spectral Normalization for Generative Adversarial Networks. In International Conference on Learning Representations.","author":"Miyato Takeru","year":"2018","unstructured":"Takeru Miyato , Toshiki Kataoka , Masanori Koyama , and Yuichi Yoshida . 2018 . Spectral Normalization for Generative Adversarial Networks. In International Conference on Learning Representations. Takeru Miyato, Toshiki Kataoka, Masanori Koyama, and Yuichi Yoshida. 2018. Spectral Normalization for Generative Adversarial Networks. In International Conference on Learning Representations."},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW.2019.00408"},{"key":"e_1_3_2_2_41_1","volume-title":"PyTorch: An Imperative Style","author":"Paszke Adam","unstructured":"Adam Paszke , Sam Gross , Francisco Massa , Adam Lerer , James Bradbury , Gregory Chanan , Trevor Killeen , Zeming Lin , Natalia Gimelshein , Luca Antiga , Alban Desmaison , Andreas Kopf , Edward Yang , Zachary DeVito , Martin Raison , Alykhan Tejani , Sasank Chilamkurthy , Benoit Steiner , Lu Fang , Junjie Bai , and Soumith Chintala . 2019. PyTorch: An Imperative Style , High-Performance Deep Learning Library . In Advances in Neural Information Processing Systems, H. Wallach, H. Larochelle, A. Beygelzimer, F. dtextquotesingle Alch\u00e9-Buc, E. Fox, and R. Garnett (Eds.), Vol. 32. Curran Associates, Inc. Adam Paszke, Sam Gross, Francisco Massa, Adam Lerer, James Bradbury, Gregory Chanan, Trevor Killeen, Zeming Lin, Natalia Gimelshein, Luca Antiga, Alban Desmaison, Andreas Kopf, Edward Yang, Zachary DeVito, Martin Raison, Alykhan Tejani, Sasank Chilamkurthy, Benoit Steiner, Lu Fang, Junjie Bai, and Soumith Chintala. 2019. PyTorch: An Imperative Style, High-Performance Deep Learning Library. In Advances in Neural Information Processing Systems, H. Wallach, H. Larochelle, A. Beygelzimer, F. dtextquotesingle Alch\u00e9-Buc, E. Fox, and R. Garnett (Eds.), Vol. 32. Curran Associates, Inc."},{"key":"e_1_3_2_2_42_1","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","author":"Pathak Deepak","unstructured":"Deepak Pathak , Philipp Krahenbuhl , Jeff Donahue , Trevor Darrell , and Alexei A. Efros . 2016. Context Encoders: Feature Learning by Inpainting . In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR). Deepak Pathak, Philipp Krahenbuhl, Jeff Donahue, Trevor Darrell, and Alexei A. Efros. 2016. Context Encoders: Feature Learning by Inpainting. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR)."},{"key":"e_1_3_2_2_43_1","volume-title":"Random Feature Attention. In International Conference on Learning Representations.","author":"Peng Hao","year":"2021","unstructured":"Hao Peng , Nikolaos Pappas , Dani Yogatama , Roy Schwartz , Noah Smith , and Lingpeng Kong . 2021 . Random Feature Attention. In International Conference on Learning Representations. Hao Peng, Nikolaos Pappas, Dani Yogatama, Roy Schwartz, Noah Smith, and Lingpeng Kong. 2021. Random Feature Attention. In International Conference on Learning Representations."},{"key":"e_1_3_2_2_44_1","volume-title":"International Conference on Learning Representations.","author":"Qin Zhen","year":"2022","unstructured":"Zhen Qin , Weixuan Sun , Hui Deng , Dongxu Li , Yunshen Wei , Baohong Lv , Junjie Yan , Lingpeng Kong , and Yiran Zhong . 2022 . cosFormer: Rethinking Softmax In Attention . In International Conference on Learning Representations. Zhen Qin, Weixuan Sun, Hui Deng, Dongxu Li, Yunshen Wei, Baohong Lv, Junjie Yan, Lingpeng Kong, and Yiran Zhong. 2022. cosFormer: Rethinking Softmax In Attention. In International Conference on Learning Representations."},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00027"},{"key":"e_1_3_2_2_46_1","volume-title":"U-Net: Convolutional Networks for Biomedical Image Segmentation","author":"Ronneberger Olaf","unstructured":"Olaf Ronneberger , Philipp Fischer , and Thomas Brox . 2015. U-Net: Convolutional Networks for Biomedical Image Segmentation . In Medical Image Computing and Computer-Assisted Intervention, Nassir Navab, Joachim Hornegger, William M. Wells, and Alejandro F. Frangi (Eds.). Springer International Publishing , Cham , 234--241. Olaf Ronneberger, Philipp Fischer, and Thomas Brox. 2015. U-Net: Convolutional Networks for Biomedical Image Segmentation. In Medical Image Computing and Computer-Assisted Intervention, Nassir Navab, Joachim Hornegger, William M. Wells, and Alejandro F. Frangi (Eds.). Springer International Publishing, Cham, 234--241."},{"key":"e_1_3_2_2_47_1","volume-title":"Glu variants improve transformer. arXiv preprint arXiv:2002.05202","author":"Shazeer Noam","year":"2020","unstructured":"Noam Shazeer . 2020. Glu variants improve transformer. arXiv preprint arXiv:2002.05202 ( 2020 ). Noam Shazeer. 2020. Glu variants improve transformer. arXiv preprint arXiv:2002.05202 (2020)."},{"key":"e_1_3_2_2_48_1","volume-title":"Very Deep Convolutional Networks for Large-Scale Image Recognition. In International Conference on Learning Representations.","author":"Simonyan Karen","year":"2015","unstructured":"Karen Simonyan and Andrew Zisserman . 2015 . Very Deep Convolutional Networks for Large-Scale Image Recognition. In International Conference on Learning Representations. Karen Simonyan and Andrew Zisserman. 2015. Very Deep Convolutional Networks for Large-Scale Image Recognition. In International Conference on Learning Representations."},{"key":"e_1_3_2_2_49_1","volume-title":"Proceedings of the 38th International Conference on Machine Learning (Proceedings of Machine Learning Research","volume":"10357","author":"Touvron Hugo","year":"2021","unstructured":"Hugo Touvron , Matthieu Cord , Matthijs Douze , Francisco Massa , Alexandre Sablayrolles , and Herve Jegou . 2021 . Training data-efficient image transformers & distillation through attention . In Proceedings of the 38th International Conference on Machine Learning (Proceedings of Machine Learning Research , Vol. 139), Marina Meila and Tong Zhang (Eds.). PMLR, 10347-- 10357 . Hugo Touvron, Matthieu Cord, Matthijs Douze, Francisco Massa, Alexandre Sablayrolles, and Herve Jegou. 2021. Training data-efficient image transformers & distillation through attention. In Proceedings of the 38th International Conference on Machine Learning (Proceedings of Machine Learning Research, Vol. 139), Marina Meila and Tong Zhang (Eds.). PMLR, 10347--10357."},{"key":"e_1_3_2_2_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01270"},{"key":"e_1_3_2_2_51_1","volume-title":"Advances in Neural Information Processing Systems","volume":"30","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani , Noam Shazeer , Niki Parmar , Jakob Uszkoreit , Llion Jones , Aidan N Gomez , \u0141ukasz Kaiser , and Illia Polosukhin . 2017 . Attention is All you Need . In Advances in Neural Information Processing Systems , Vol. 30 . Curran Associates, Inc. Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is All you Need. In Advances in Neural Information Processing Systems, Vol. 30. Curran Associates, Inc."},{"key":"e_1_3_2_2_52_1","volume-title":"Garnett (Eds.)","volume":"29","author":"Veit Andreas","year":"2016","unstructured":"Andreas Veit , Michael J Wilber , and Serge Belongie . 2016 . Residual Networks Behave Like Ensembles of Relatively Shallow Networks. In Advances in Neural Information Processing Systems, D. Lee, M. Sugiyama, U. Luxburg, I. Guyon, and R . Garnett (Eds.) , Vol. 29 . Curran Associates, Inc. Andreas Veit, Michael J Wilber, and Serge Belongie. 2016. Residual Networks Behave Like Ensembles of Relatively Shallow Networks. In Advances in Neural Information Processing Systems, D. Lee, M. Sugiyama, U. Luxburg, I. Guyon, and R. Garnett (Eds.), Vol. 29. Curran Associates, Inc."},{"key":"e_1_3_2_2_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00465"},{"key":"e_1_3_2_2_54_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2019\/520"},{"key":"e_1_3_2_2_55_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00061"},{"key":"e_1_3_2_2_56_1","volume-title":"Uformer: A general u-shaped transformer for image restoration. arXiv preprint arXiv:2106.03106","author":"Wang Zhendong","year":"2021","unstructured":"Zhendong Wang , Xiaodong Cun , Jianmin Bao , and Jianzhuang Liu . 2021 a. Uformer: A general u-shaped transformer for image restoration. arXiv preprint arXiv:2106.03106 (2021). Zhendong Wang, Xiaodong Cun, Jianmin Bao, and Jianzhuang Liu. 2021a. Uformer: A general u-shaped transformer for image restoration. arXiv preprint arXiv:2106.03106 (2021)."},{"key":"e_1_3_2_2_57_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00895"},{"key":"e_1_3_2_2_58_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2010.2042098"},{"key":"e_1_3_2_2_59_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01264-9_1"},{"key":"e_1_3_2_2_60_1","volume-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","author":"Yu Jiahui","unstructured":"Jiahui Yu , Zhe Lin , Jimei Yang , Xiaohui Shen , Xin Lu , and Thomas S. Huang . 2018. Generative Image Inpainting With Contextual Attention . In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR). Jiahui Yu, Zhe Lin, Jimei Yang, Xiaohui Shen, Xin Lu, and Thomas S. Huang. 2018. Generative Image Inpainting With Contextual Attention. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR)."},{"key":"e_1_3_2_2_61_1","volume-title":"Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV).","author":"Yu Jiahui","unstructured":"Jiahui Yu , Zhe Lin , Jimei Yang , Xiaohui Shen , Xin Lu , and Thomas S. Huang . 2019. Free-Form Image Inpainting With Gated Convolution . In Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV). Jiahui Yu, Zhe Lin, Jimei Yang, Xiaohui Shen, Xin Lu, and Thomas S. Huang. 2019. Free-Form Image Inpainting With Gated Convolution. In Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV)."},{"key":"e_1_3_2_2_62_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474085.3475436"},{"key":"e_1_3_2_2_63_1","volume-title":"Fahad Shahbaz Khan, and Ming-Hsuan Yang","author":"Zamir Syed Waqas","year":"2021","unstructured":"Syed Waqas Zamir , Aditya Arora , Salman Khan , Munawar Hayat , Fahad Shahbaz Khan, and Ming-Hsuan Yang . 2021 . Restormer : Efficient Transformer for High-Resolution Image Restoration . arXiv preprint arXiv:2111.09881 (2021). Syed Waqas Zamir, Aditya Arora, Salman Khan, Munawar Hayat, Fahad Shahbaz Khan, and Ming-Hsuan Yang. 2021. Restormer: Efficient Transformer for High-Resolution Image Restoration. arXiv preprint arXiv:2111.09881 (2021)."},{"key":"e_1_3_2_2_64_1","volume-title":"European Conference on Computer Vision, Andrea Vedaldi, Horst Bischof, Thomas Brox, and Jan-Michael Frahm (Eds.)","author":"Zeng Yanhong","unstructured":"Yanhong Zeng , Jianlong Fu , and Hongyang Chao . 2020. Learning Joint Spatial-Temporal Transformations for Video Inpainting . In European Conference on Computer Vision, Andrea Vedaldi, Horst Bischof, Thomas Brox, and Jan-Michael Frahm (Eds.) . Springer International Publishing . Yanhong Zeng, Jianlong Fu, and Hongyang Chao. 2020. Learning Joint Spatial-Temporal Transformations for Video Inpainting. In European Conference on Computer Vision, Andrea Vedaldi, Horst Bischof, Thomas Brox, and Jan-Michael Frahm (Eds.). Springer International Publishing."},{"key":"e_1_3_2_2_65_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00158"},{"key":"e_1_3_2_2_66_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00153"},{"key":"e_1_3_2_2_67_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01122"},{"key":"e_1_3_2_2_68_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2017.2723009"},{"key":"e_1_3_2_2_69_1","volume-title":"Proceedings of the IEEE International Conference on Computer Vision (ICCV).","author":"Zhu Jun-Yan","unstructured":"Jun-Yan Zhu , Taesung Park , Phillip Isola , and Alexei A. Efros . 2017. Unpaired Image-To-Image Translation Using Cycle-Consistent Adversarial Networks . In Proceedings of the IEEE International Conference on Computer Vision (ICCV). Jun-Yan Zhu, Taesung Park, Phillip Isola, and Alexei A. Efros. 2017. Unpaired Image-To-Image Translation Using Cycle-Consistent Adversarial Networks. In Proceedings of the IEEE International Conference on Computer Vision (ICCV)."}],"event":{"name":"MM '22: The 30th ACM International Conference on Multimedia","location":"Lisboa Portugal","acronym":"MM '22","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 30th ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3503161.3548446","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3503161.3548446","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T17:49:17Z","timestamp":1750182557000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3503161.3548446"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,10,10]]},"references-count":69,"alternative-id":["10.1145\/3503161.3548446","10.1145\/3503161"],"URL":"https:\/\/doi.org\/10.1145\/3503161.3548446","relation":{},"subject":[],"published":{"date-parts":[[2022,10,10]]},"assertion":[{"value":"2022-10-10","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}