{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T15:18:13Z","timestamp":1778080693085,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":41,"publisher":"ACM","license":[{"start":{"date-parts":[[2021,10,17]],"date-time":"2021-10-17T00:00:00Z","timestamp":1634428800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"National Key Research and Development Program of China","award":["2018AAA0102200"],"award-info":[{"award-number":["2018AAA0102200"]}]},{"name":"National Natural Science Foundation of China","award":["61772116"],"award-info":[{"award-number":["61772116"]}]},{"name":"National Natural Science Foundation of China","award":["61872064"],"award-info":[{"award-number":["61872064"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2021,10,17]]},"DOI":"10.1145\/3474085.3475326","type":"proceedings-article","created":{"date-parts":[[2021,10,18]],"date-time":"2021-10-18T05:04:15Z","timestamp":1634533455000},"page":"1784-1792","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":14,"title":["Fully Functional Image Manipulation Using Scene Graphs in A Bounding-Box Free Way"],"prefix":"10.1145","author":[{"given":"Sitong","family":"Su","sequence":"first","affiliation":[{"name":"University of Electronic Science and Technology of China, Cheng Du, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lianli","family":"Gao","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, Cheng Du, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Junchen","family":"Zhu","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, Cheng Du, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jie","family":"Shao","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, Cheng Du, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jingkuan","family":"Song","sequence":"additional","affiliation":[{"name":"University of Electronic Science and Technology of China, Cheng Du, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2021,10,17]]},"reference":[{"key":"e_1_3_2_1_1_1","first-page":"382","article-title":"SPICE: Semantic Propositional Image Caption Evaluation","volume":"9909","author":"Anderson Peter","year":"2016","journal-title":"ECCV"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"crossref","unstructured":"Oron Ashual and Lior Wolf. 2019. Specifying Object Attributes and Relations in Interactive Scene Generation. In ICCV. 4560--4568.  Oron Ashual and Lior Wolf. 2019. Specifying Object Attributes and Relations in Interactive Scene Generation. In ICCV. 4560--4568.","DOI":"10.1109\/ICCV.2019.00466"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"crossref","unstructured":"Holger Caesar Jasper R. R. Uijlings and Vittorio Ferrari. 2018. COCO-Stuff: Thing and Stuff Classes in Context. In CVPR. 1209--1218.  Holger Caesar Jasper R. R. Uijlings and Vittorio Ferrari. 2018. COCO-Stuff: Thing and Stuff Classes in Context. In CVPR. 1209--1218.","DOI":"10.1109\/CVPR.2018.00132"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"crossref","unstructured":"Wengling Chen and James Hays. 2018. SketchyGAN: Towards Diverse and Realistic Sketch to Image Synthesis. In CVPR. 9416--9425.  Wengling Chen and James Hays. 2018. SketchyGAN: Towards Diverse and Realistic Sketch to Image Synthesis. In CVPR. 9416--9425.","DOI":"10.1109\/CVPR.2018.00981"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"crossref","unstructured":"Bo Dai Yuqi Zhang and Dahua Lin. 2017. Detecting Visual Relationships with Deep Relational Networks. In CVPR. 3298--3308.  Bo Dai Yuqi Zhang and Dahua Lin. 2017. Detecting Visual Relationships with Deep Relational Networks. In CVPR. 3298--3308.","DOI":"10.1109\/CVPR.2017.352"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"crossref","unstructured":"Helisa Dhamo Azade Farshad Iro Laina Nassir Navab Gregory D. Hager Federico Tombari and Christian Rupprecht. 2020. Semantic Image Manipulation Using Scene Graphs. In CVPR. 5212--5221.  Helisa Dhamo Azade Farshad Iro Laina Nassir Navab Gregory D. Hager Federico Tombari and Christian Rupprecht. 2020. Semantic Image Manipulation Using Scene Graphs. In CVPR. 5212--5221.","DOI":"10.1109\/CVPR42600.2020.00526"},{"key":"e_1_3_2_1_7_1","unstructured":"Chengying Gao Qi Liu Qi Xu Limin Wang Jianzhuang Liu and Changqing Zou. 2020. SketchyCOCO: Image Generation From Freehand Scene Sketches. In CVPR. 5173--5182.  Chengying Gao Qi Liu Qi Xu Limin Wang Jianzhuang Liu and Changqing Zou. 2020. SketchyCOCO: Image Generation From Freehand Scene Sketches. In CVPR. 5173--5182."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"crossref","unstructured":"Arnab Ghosh Richard Zhang Puneet K. Dokania Oliver Wang Alexei A. Efros Philip H. S. Torr and Eli Shechtman. 2019. Interactive Sketch & Fill: Multiclass Sketch-to-Image Translation. In ICCV .  Arnab Ghosh Richard Zhang Puneet K. Dokania Oliver Wang Alexei A. Efros Philip H. S. Torr and Eli Shechtman. 2019. Interactive Sketch & Fill: Multiclass Sketch-to-Image Translation. In ICCV .","DOI":"10.1109\/ICCV.2019.00126"},{"key":"e_1_3_2_1_9_1","unstructured":"Kaiming He Xiangyu Zhang Shaoqing Ren and Jian Sun. 2016. Deep Residual Learning for Image Recognition. In CVPR. 770--778.  Kaiming He Xiangyu Zhang Shaoqing Ren and Jian Sun. 2016. Deep Residual Learning for Image Recognition. In CVPR. 770--778."},{"key":"e_1_3_2_1_10_1","first-page":"210","article-title":"Learning Canonical Representations for Scene Graph to Image Generation","volume":"12371","author":"Herzig Roei","year":"2020","journal-title":"ECCV"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.5555\/3295222.3295408"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01219-9_11"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"crossref","unstructured":"Seong Jae Hwang Sathya N. Ravi Zirui Tao Hyunwoo J. Kim Maxwell D. Collins and Vikas Singh. 2018. Tensorize Factorize and Regularize: Robust Visual Relationship Learning. In CVPR. 1014--1023.  Seong Jae Hwang Sathya N. Ravi Zirui Tao Hyunwoo J. Kim Maxwell D. Collins and Vikas Singh. 2018. Tensorize Factorize and Regularize: Robust Visual Relationship Learning. In CVPR. 1014--1023.","DOI":"10.1109\/CVPR.2018.00112"},{"key":"e_1_3_2_1_14_1","volume-title":"Efros","author":"Isola Phillip","year":"2017"},{"key":"e_1_3_2_1_15_1","volume-title":"Scene Graph to Image Generation with Contextualized Object Layout Refinement. CoRR","author":"Ivgi Maor","year":"2020"},{"key":"e_1_3_2_1_16_1","unstructured":"Youngjoo Jo and Jongyoul Park. 2019. SC-FEGAN: Face Editing Generative Adversarial Network With User's Sketch and Color. In ICCV. 1745--1753.  Youngjoo Jo and Jongyoul Park. 2019. SC-FEGAN: Face Editing Generative Adversarial Network With User's Sketch and Color. In ICCV. 1745--1753."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"crossref","unstructured":"Justin Johnson Agrim Gupta and Li Fei-Fei. 2018. Image Generation From Scene Graphs. In CVPR. 1219--1228.  Justin Johnson Agrim Gupta and Li Fei-Fei. 2018. Image Generation From Scene Graphs. In CVPR. 1219--1228.","DOI":"10.1109\/CVPR.2018.00133"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"crossref","unstructured":"Justin Johnson Ranjay Krishna Michael Stark Li-Jia Li David A. Shamma Michael S. Bernstein and Fei-Fei Li. 2015. Image retrieval using scene graphs. In CVPR. 3668--3678.  Justin Johnson Ranjay Krishna Michael Stark Li-Jia Li David A. Shamma Michael S. Bernstein and Fei-Fei Li. 2015. Image retrieval using scene graphs. In CVPR. 3668--3678.","DOI":"10.1109\/CVPR.2015.7298990"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-016-0981-7"},{"key":"e_1_3_2_1_20_1","first-page":"36","article-title":"Diverse Image-to-Image Translation via Disentangled Representations","volume":"11205","author":"Lee Hsin-Ying","year":"2018","journal-title":"ECCV"},{"key":"e_1_3_2_1_21_1","unstructured":"Bowen Li Xiaojuan Qi Thomas Lukasiewicz and Philip H. S. Torr. 2020 a. ManiGAN: Text-Guided Image Manipulation. In CVPR. 7877--7886.  Bowen Li Xiaojuan Qi Thomas Lukasiewicz and Philip H. S. Torr. 2020 a. ManiGAN: Text-Guided Image Manipulation. In CVPR. 7877--7886."},{"key":"e_1_3_2_1_22_1","unstructured":"Bowen Li Xiaojuan Qi Philip H. S. Torr and Thomas Lukasiewicz. 2020 b. Lightweight Generative Adversarial Networks for Text-Guided Image Manipulation. In NeurIPS .  Bowen Li Xiaojuan Qi Philip H. S. Torr and Thomas Lukasiewicz. 2020 b. Lightweight Generative Adversarial Networks for Text-Guided Image Manipulation. In NeurIPS ."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3394171.3414027"},{"key":"e_1_3_2_1_24_1","volume-title":"PULSE: Self-Supervised Photo Upsampling via Latent Space Exploration of Generative Models. In CVPR. 2434--2442.","author":"Menon Sachit","year":"2020"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.5555\/3326943.3326948"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"crossref","unstructured":"Taesung Park Ming-Yu Liu Ting-Chun Wang and Jun-Yan Zhu. 2019. Semantic Image Synthesis With Spatially-Adaptive Normalization. In CVPR. 2337--2346.  Taesung Park Ming-Yu Liu Ting-Chun Wang and Jun-Yan Zhu. 2019. Semantic Image Synthesis With Spatially-Adaptive Normalization. In CVPR. 2337--2346.","DOI":"10.1109\/CVPR.2019.00244"},{"key":"e_1_3_2_1_27_1","first-page":"482","article-title":"Controlling Style and Semantics in Weakly-Supervised Image Generation","volume":"12351","author":"Pavllo Dario","year":"2020","journal-title":"ECCV"},{"key":"e_1_3_2_1_28_1","volume-title":"MRA-Net: Improving VQA via Multi-modal Relation Attention Network","author":"Peng Liang","year":"2020"},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"crossref","unstructured":"Wei Sun and Tianfu Wu. 2019. Image Synthesis From Reconfigurable Layout and Style. In ICCV. 10530--10539.  Wei Sun and Tianfu Wu. 2019. Image Synthesis From Reconfigurable Layout and Style. In ICCV. 10530--10539.","DOI":"10.1109\/ICCV.2019.01063"},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"crossref","unstructured":"Kaihua Tang Yulei Niu Jianqiang Huang Jiaxin Shi and Hanwang Zhang. 2020. Unbiased Scene Graph Generation From Biased Training. In CVPR. 3713--3722.  Kaihua Tang Yulei Niu Jianqiang Huang Jiaxin Shi and Hanwang Zhang. 2020. Unbiased Scene Graph Generation From Biased Training. In CVPR. 3713--3722.","DOI":"10.1109\/CVPR42600.2020.00377"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"crossref","unstructured":"Kaihua Tang Hanwang Zhang Baoyuan Wu Wenhan Luo and Wei Liu. 2019. Learning to Compose Dynamic Tree Structures for Visual Contexts. In CVPR. 6619--6628.  Kaihua Tang Hanwang Zhang Baoyuan Wu Wenhan Luo and Wei Liu. 2019. Learning to Compose Dynamic Tree Structures for Visual Contexts. In CVPR. 6619--6628.","DOI":"10.1109\/CVPR.2019.00678"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"crossref","unstructured":"Xintao Wang Ke Yu Chao Dong and Chen Change Loy. 2018. Recovering Realistic Texture in Image Super-Resolution by Deep Spatial Feature Transform. In CVPR. 606--615.  Xintao Wang Ke Yu Chao Dong and Chen Change Loy. 2018. Recovering Realistic Texture in Image Super-Resolution by Deep Spatial Feature Transform. In CVPR. 606--615.","DOI":"10.1109\/CVPR.2018.00070"},{"key":"e_1_3_2_1_33_1","unstructured":"Yuxin Wu Alexander Kirillov Francisco Massa Wan-Yen Lo and Ross Girshick. 2019. Detectron2. https:\/\/github.com\/facebookresearch\/detectron2.  Yuxin Wu Alexander Kirillov Francisco Massa Wan-Yen Lo and Ross Girshick. 2019. Detectron2. https:\/\/github.com\/facebookresearch\/detectron2."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"crossref","unstructured":"Tao Xu Pengchuan Zhang Qiuyuan Huang Han Zhang Zhe Gan Xiaolei Huang and Xiaodong He. 2018. AttnGAN: Fine-Grained Text to Image Generation With Attentional Generative Adversarial Networks. In CVPR. 1316--1324.  Tao Xu Pengchuan Zhang Qiuyuan Huang Han Zhang Zhe Gan Xiaolei Huang and Xiaodong He. 2018. AttnGAN: Fine-Grained Text to Image Generation With Attentional Generative Adversarial Networks. In CVPR. 1316--1324.","DOI":"10.1109\/CVPR.2018.00143"},{"key":"e_1_3_2_1_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/TNNLS.2020.2986029"},{"key":"e_1_3_2_1_36_1","first-page":"690","article-title":"Graph R-CNN for Scene Graph Generation","volume":"11205","author":"Yang Jianwei","year":"2018","journal-title":"ECCV"},{"key":"e_1_3_2_1_37_1","unstructured":"Zili Yi Qiang Tang Shekoofeh Azizi Daesik Jang and Zhan Xu. 2020. Contextual Residual Aggregation for Ultra High-Resolution Image Inpainting. In CVPR. 7505--7514.  Zili Yi Qiang Tang Shekoofeh Azizi Daesik Jang and Zhan Xu. 2020. Contextual Residual Aggregation for Ultra High-Resolution Image Inpainting. In CVPR. 7505--7514."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"crossref","unstructured":"Han Zhang Tao Xu and Hongsheng Li. 2017. StackGAN: Text to Photo-Realistic Image Synthesis with Stacked Generative Adversarial Networks. In ICCV. 5908--5916.  Han Zhang Tao Xu and Hongsheng Li. 2017. StackGAN: Text to Photo-Realistic Image Synthesis with Stacked Generative Adversarial Networks. In ICCV. 5908--5916.","DOI":"10.1109\/ICCV.2017.629"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"crossref","unstructured":"Richard Zhang Phillip Isola Alexei A. Efros Eli Shechtman and Oliver Wang. 2018. The Unreasonable Effectiveness of Deep Features as a Perceptual Metric. In CVPR. 586--595.  Richard Zhang Phillip Isola Alexei A. Efros Eli Shechtman and Oliver Wang. 2018. The Unreasonable Effectiveness of Deep Features as a Perceptual Metric. In CVPR. 586--595.","DOI":"10.1109\/CVPR.2018.00068"},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"crossref","unstructured":"Bo Zhao Lili Meng Weidong Yin and Leonid Sigal. 2019. Image Generation From Layout. In CVPR. 8584--8593.  Bo Zhao Lili Meng Weidong Yin and Leonid Sigal. 2019. Image Generation From Layout. In CVPR. 8584--8593.","DOI":"10.1109\/CVPR.2019.00878"},{"key":"e_1_3_2_1_41_1","volume-title":"Efros","author":"Zhu Jun-Yan","year":"2017"}],"event":{"name":"MM '21: ACM Multimedia Conference","location":"Virtual Event China","acronym":"MM '21","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 29th ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3474085.3475326","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3474085.3475326","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T20:49:18Z","timestamp":1750193358000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3474085.3475326"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,10,17]]},"references-count":41,"alternative-id":["10.1145\/3474085.3475326","10.1145\/3474085"],"URL":"https:\/\/doi.org\/10.1145\/3474085.3475326","relation":{},"subject":[],"published":{"date-parts":[[2021,10,17]]},"assertion":[{"value":"2021-10-17","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}