{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,30]],"date-time":"2026-01-30T05:26:36Z","timestamp":1769750796997,"version":"3.49.0"},"publisher-location":"New York, NY, USA","reference-count":65,"publisher":"ACM","license":[{"start":{"date-parts":[[2021,10,17]],"date-time":"2021-10-17T00:00:00Z","timestamp":1634428800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Ministry of Science and Technology of the People's Republic of China","award":["the National Key Research and Development Program of China under Grant No. 2017YFA0700800"],"award-info":[{"award-number":["the National Key Research and Development Program of China under Grant No. 2017YFA0700800"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2021,10,17]]},"DOI":"10.1145\/3474085.3475426","type":"proceedings-article","created":{"date-parts":[[2021,10,18]],"date-time":"2021-10-18T04:52:26Z","timestamp":1634532746000},"page":"2529-2538","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":25,"title":["Learning Contextual Transformer Network for Image Inpainting"],"prefix":"10.1145","author":[{"given":"Ye","family":"Deng","sequence":"first","affiliation":[{"name":"Xi'an Jiaotong University, Xi'an, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Siqi","family":"Hui","sequence":"additional","affiliation":[{"name":"Xi'an Jiaotong University, Xi'an, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sanping","family":"Zhou","sequence":"additional","affiliation":[{"name":"Xi'an Jiaotong University &amp; Shunan Academy of Artificial Intelligence, Xi'an &amp; Ningbo, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Deyu","family":"Meng","sequence":"additional","affiliation":[{"name":"Xi'an Jiaotong University, Xi'an, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jinjun","family":"Wang","sequence":"additional","affiliation":[{"name":"Xi'an Jiaotong University, Xi'an, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2021,10,17]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/83.935036"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/1576246.1531330"},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/344779.344972"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2003.815261"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2004.833105"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/2185520.2185578"},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"e_1_3_2_2_9_1","volume-title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale. arXiv preprint arXiv:2010.11929","author":"Dosovitskiy Alexey","year":"2020"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.265"},{"key":"e_1_3_2_2_11_1","volume-title":"Proceedings of the fourteenth international conference on artificial intelligence and statistics. JMLR Workshop and Conference Proceedings, 315--323","author":"Glorot Xavier","year":"2011"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.5555\/2969033.2969125"},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_2_14_1","volume-title":"Proceedings of the European Conference on Computer Vision .","author":"Wei Huang Hongyu Liu Yibing Song","year":"2020"},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3072959.3073659"},{"key":"e_1_3_2_2_16_1","volume-title":"Proceedings of the AAAI Conference on Artificial Intelligence","volume":"34","author":"Jie Yang Yong Shi","year":"2020"},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46475-6_43"},{"key":"e_1_3_2_2_18_1","volume-title":"Progressive growing of gans for improved quality, stability, and variation. arXiv preprint arXiv:1710.10196","author":"Karras Tero","year":"2017"},{"key":"e_1_3_2_2_19_1","volume-title":"Adam: A method for stochastic optimization. arXiv preprint arXiv:1412.6980","author":"Kingma Diederik P","year":"2014"},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.5555\/946247.946661"},{"key":"e_1_3_2_2_21_1","volume-title":"Progressive Reconstruction of Visual Structure for Image Inpainting. In The IEEE International Conference on Computer Vision (ICCV) .","author":"Li Jingyuan","year":"2019"},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00778"},{"key":"e_1_3_2_2_23_1","volume-title":"Localvit: Bringing locality to vision transformers. arXiv preprint arXiv:2104.05707","author":"Li Yawei","year":"2021"},{"key":"e_1_3_2_2_24_1","volume-title":"Image Inpainting for Irregular Holes Using Partial Convolutions. In The European Conference on Computer Vision (ECCV) .","author":"Liu Guilin","year":"2018"},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00427"},{"key":"e_1_3_2_2_26_1","volume-title":"Coherent Semantic Attention for Image Inpainting. In IEEE International Conference on Computer Vision (ICCV) .","author":"Liu Hongyu","year":"2019"},{"key":"e_1_3_2_2_27_1","volume-title":"Swin transformer: Hierarchical vision transformer using shifted windows. arXiv preprint arXiv:2103.14030","author":"Liu Ze","year":"2021"},{"key":"e_1_3_2_2_28_1","volume-title":"2020 a. Pyramid attention networks for image restoration. arXiv preprint arXiv:2004.13824","author":"Mei Yiqun","year":"2020"},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00573"},{"key":"e_1_3_2_2_30_1","volume-title":"Spectral normalization for generative adversarial networks. arXiv preprint arXiv:1802.05957","author":"Miyato Takeru","year":"2018"},{"key":"e_1_3_2_2_31_1","unstructured":"Kamyar Nazeri Eric Ng Tony Joseph Faisal Qureshi and Mehran Ebrahimi. 2019. EdgeConnect: Generative Image Inpainting with Adversarial Edge Learning. arXiv preprint .  Kamyar Nazeri Eric Ng Tony Joseph Faisal Qureshi and Mehran Ebrahimi. 2019. EdgeConnect: Generative Image Inpainting with Adversarial Edge Learning. arXiv preprint ."},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"publisher","DOI":"10.5555\/3454287.3455008"},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.278"},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00027"},{"key":"e_1_3_2_2_35_1","volume-title":"Very deep convolutional networks for large-scale image recognition. arXiv preprint arXiv:1409.1556","author":"Simonyan Karen","year":"2014"},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01216-8_1"},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/1186822.1073274"},{"key":"e_1_3_2_2_38_1","volume-title":"International Conference on Machine Learning. PMLR, 10347--10357","author":"Touvron Hugo","year":"2021"},{"key":"e_1_3_2_2_39_1","volume-title":"Well-Read Students Learn Better: On the Importance of Pre-training Compact Models. arXiv preprint arXiv:1908.08962v2","author":"Turc Iulia","year":"2019"},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.5555\/3295222.3295349"},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"publisher","DOI":"10.5555\/3367471.3367562"},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00813"},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58595-2_45"},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.5555\/3326943.3326974"},{"key":"e_1_3_2_2_45_1","volume-title":"Image Inpainting With Learnable Bidirectional Attention Maps. In The IEEE International Conference on Computer Vision (ICCV) .","author":"Xie Chaohao","year":"2019"},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01264-9_1"},{"key":"e_1_3_2_2_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.434"},{"key":"e_1_3_2_2_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00583"},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.728"},{"key":"e_1_3_2_2_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00753"},{"key":"e_1_3_2_2_51_1","volume-title":"International Conference on Learning Representations (ICLR) .","author":"Yu Fisher","year":"2016"},{"key":"e_1_3_2_2_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00577"},{"key":"e_1_3_2_2_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00457"},{"key":"e_1_3_2_2_54_1","volume-title":"Jiashi Feng, and Shuicheng Yan.","author":"Yuan Li","year":"2021"},{"key":"e_1_3_2_2_55_1","volume-title":"Variational Image Restoration Network. arXiv preprint arXiv:2008.10796","author":"Yue Zongsheng","year":"2020"},{"key":"e_1_3_2_2_56_1","volume-title":"Learning Joint Spatial-Temporal Transformations for Video Inpainting. In European Conference on Computer Vision. Springer, 528--543","author":"Zeng Yanhong","year":"2020"},{"key":"e_1_3_2_2_57_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00158"},{"key":"e_1_3_2_2_58_1","volume-title":"Learning Pyramid-Context Encoder Network for High-Quality Image Inpainting. In The IEEE Conference on Computer Vision and Pattern Recognition (CVPR). 1486--1494","author":"Zeng Yanhong","year":"2019"},{"key":"e_1_3_2_2_59_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58529-7_1"},{"key":"e_1_3_2_2_60_1","volume-title":"International conference on machine learning. PMLR, 7354--7363","author":"Zhang Han","year":"2019"},{"key":"e_1_3_2_2_61_1","doi-asserted-by":"publisher","DOI":"10.1145\/3240508.3240625"},{"key":"e_1_3_2_2_62_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00578"},{"key":"e_1_3_2_2_63_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00153"},{"key":"e_1_3_2_2_64_1","volume-title":"Places: A 10 million Image Database for Scene Recognition","author":"Zhou Bolei","year":"2017"},{"key":"e_1_3_2_2_65_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.244"}],"event":{"name":"MM '21: ACM Multimedia Conference","location":"Virtual Event China","acronym":"MM '21","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 29th ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3474085.3475426","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3474085.3475426","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T20:48:33Z","timestamp":1750193313000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3474085.3475426"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,10,17]]},"references-count":65,"alternative-id":["10.1145\/3474085.3475426","10.1145\/3474085"],"URL":"https:\/\/doi.org\/10.1145\/3474085.3475426","relation":{},"subject":[],"published":{"date-parts":[[2021,10,17]]},"assertion":[{"value":"2021-10-17","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}