{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,22]],"date-time":"2026-07-22T23:13:15Z","timestamp":1784761995309,"version":"3.55.0"},"reference-count":57,"publisher":"Association for Computing Machinery (ACM)","issue":"4","license":[{"start":{"date-parts":[[2017,7,11]],"date-time":"2017-07-11T00:00:00Z","timestamp":1499731200000},"content-version":"vor","delay-in-days":365,"URL":"http:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["1149853"],"award-info":[{"award-number":["1149853"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":["ACM Trans. Graph."],"published-print":{"date-parts":[[2016,7,11]]},"abstract":"<jats:p>\n                    We present the\n                    <jats:italic toggle=\"yes\">Sketchy database<\/jats:italic>\n                    , the first large-scale collection of sketch-photo pairs. We ask crowd workers to sketch particular photographic objects sampled from 125 categories and acquire 75,471 sketches of 12,500 objects. The Sketchy database gives us fine-grained associations between particular photos and sketches, and we use this to train cross-domain convolutional networks which embed sketches and photographs in a common feature space. We use our database as a benchmark for fine-grained retrieval and show that our learned representation significantly outperforms both hand-crafted features as well as deep features trained for sketch or photo classification. Beyond image retrieval, we believe the Sketchy database opens up new opportunities for sketch and image understanding and synthesis.\n                  <\/jats:p>","DOI":"10.1145\/2897824.2925954","type":"journal-article","created":{"date-parts":[[2016,7,11]],"date-time":"2016-07-11T12:04:33Z","timestamp":1468238673000},"page":"1-12","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":542,"title":["The sketchy database"],"prefix":"10.1145","volume":"35","author":[{"given":"Patsorn","family":"Sangkloy","sequence":"first","affiliation":[{"name":"Georgia Institute of Technology"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Nathan","family":"Burnell","sequence":"additional","affiliation":[{"name":"Brown University"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Cusuh","family":"Ham","sequence":"additional","affiliation":[{"name":"Georgia Institute of Technology"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"James","family":"Hays","sequence":"additional","affiliation":[{"name":"Georgia Institute of Technology"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2016,7,11]]},"reference":[{"key":"e_1_2_2_1_1","doi-asserted-by":"crossref","unstructured":"Antol S. Zitnick C. L. and Parikh D. 2014. Zero-Shot Learning via Visual Abstraction. In ECCV.","DOI":"10.1007\/978-3-319-10593-2_27"},{"key":"e_1_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW.2013.80"},{"key":"e_1_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/2766959"},{"key":"e_1_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/2461912.2461964"},{"key":"e_1_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.0803390105"},{"key":"e_1_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1177\/0956797612465439"},{"key":"e_1_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2011.5995460"},{"key":"e_1_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2013.46"},{"key":"e_1_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/1661412.1618470"},{"key":"e_1_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/TVCG.2012.148"},{"key":"e_1_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2005.202"},{"key":"e_1_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/1360612.1360687"},{"key":"e_1_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/34.574790"},{"key":"e_1_2_2_14_1","unstructured":"Dosovitskiy A. Springenberg J. T. and Brox T. 2014. Learning to generate chairs with convolutional neural networks. CoRR abs\/1411.5928."},{"key":"e_1_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.cag.2010.07.002"},{"key":"e_1_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/TVCG.2010.266"},{"key":"e_1_2_2_17_1","doi-asserted-by":"publisher","unstructured":"Eitz M. Richter R. Hildebrand K. Boubekeur T. and Alexa M. 2011. Photosketcher: interactive sketch-based image synthesis. IEEE Computer Graphics and Applications. 10.1109\/MCG.2011.67","DOI":"10.1109\/MCG.2011.67"},{"key":"e_1_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/2185520.2185540"},{"key":"e_1_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/2185520.2185527"},{"key":"e_1_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-009-0275-4"},{"key":"e_1_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2009.167"},{"key":"e_1_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1111\/j.0956-7976.2005.00796.x"},{"key":"e_1_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2006.100"},{"key":"e_1_2_2_24_1","volume-title":"Computer Vision and Pattern Recognition (CVPR), 2015 IEEE Conference on, 3279--3286","author":"Han X.","unstructured":"Han, X., Leung, T., Jia, Y., Sukthankar, R., and Berg, A. 2015. Matchnet: Unifying feature and metric learning for patch-based matching. In Computer Vision and Pattern Recognition (CVPR), 2015 IEEE Conference on, 3279--3286."},{"key":"e_1_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.cviu.2013.02.005"},{"key":"e_1_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/218380.218454"},{"key":"e_1_2_2_27_1","volume-title":"Caffe: Convolutional architecture for fast feature embedding. arXiv preprint arXiv:1408.5093.","author":"Jia Y.","year":"2014","unstructured":"Jia, Y., Shelhamer, E., Donahue, J., Karayev, S., Long, J., Girshick, R., Guadarrama, S., and Darrell, T. 2014. Caffe: Convolutional architecture for fast feature embedding. arXiv preprint arXiv:1408.5093."},{"key":"e_1_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/2642918.2647399"},{"key":"e_1_2_2_29_1","volume-title":"Pattern Recognition, 1992.","author":"Kato T.","unstructured":"Kato, T., Kurita, T., Otsu, N., and Hirata, K. 1992. A sketch retrieval method for full color image database-query by visual example. In Pattern Recognition, 1992. Vol. I. Conference A: Computer Vision and Applications, Proceedings., 11th IAPR International Conference on, 530--533."},{"key":"e_1_2_2_30_1","volume-title":"26th Annual Conference on Neural Information Processing Systems (NIPS), 1106--1114","author":"Krizhevsky A.","unstructured":"Krizhevsky, A., Sutskever, I., and Hinton, G. E. 2012. Imagenet classification with deep convolutional neural networks. In 26th Annual Conference on Neural Information Processing Systems (NIPS), 1106--1114."},{"key":"e_1_2_2_31_1","doi-asserted-by":"crossref","unstructured":"Lee D. and Chun M. M. What are the units of visual short-term memory objects or spatial locations? Perception & Psychophysics 63 2 253--257.","DOI":"10.3758\/BF03194466"},{"key":"e_1_2_2_32_1","volume-title":"British Machine Vision Conference (BMVC).","author":"Li Y.","unstructured":"Li, Y., Hospedales, T. M., Song, Y.-Z., and Gong, S. 2014. Fine-grained sketch-based image retrieval by matching deformable part models. In British Machine Vision Conference (BMVC)."},{"key":"e_1_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.1145\/2816795.2818071"},{"key":"e_1_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/2461912.2462016"},{"key":"e_1_2_2_35_1","doi-asserted-by":"crossref","unstructured":"Lin T. Maire M. Belongie S. J. Bourdev L. D. Girshick R. B. Hays J. Perona P. Ramanan D. Doll\u00e1r P. and Zitnick C. L. 2014. Microsoft COCO: common objects in context. CoRR abs\/1405.0312.","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"e_1_2_2_36_1","volume-title":"The IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","author":"Lin T.-Y.","unstructured":"Lin, T.-Y., Cui, Y., Belongie, S., and Hays, J. 2015. Learning deep representations for ground-to-aerial geolocalization. In The IEEE Conference on Computer Vision and Pattern Recognition (CVPR)."},{"key":"e_1_2_2_37_1","volume-title":"March 20, 2015.","author":"Mainelli T.","unstructured":"Mainelli, T., Chau, M., Reith, R., and Shirer, M., 2015. Idc worldwide quarterly smart connected device tracker. http:\/\/www.idc.com\/getdoc.jsp?containerId=prUS25500515, March 20, 2015."},{"key":"e_1_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2001.937655"},{"key":"e_1_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.1037\/a0035257"},{"key":"e_1_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-015-0816-y"},{"key":"e_1_2_2_41_1","first-page":"1","article-title":"Sketch based image retrieval using learned keyshapes (lks)","volume":"164","author":"Saavedra J. M.","year":"2015","unstructured":"Saavedra, J. M., and Barrios, J. M. 2015. Sketch based image retrieval using learned keyshapes (lks). In Proceedings of the British Machine Vision Conference (BMVC), 164.1--164.11.","journal-title":"Proceedings of the British Machine Vision Conference (BMVC)"},{"key":"e_1_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/2661229.2661231"},{"key":"e_1_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.1016\/S0031-3203(96)00108-2"},{"key":"e_1_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.1145\/2070781.2024188"},{"key":"e_1_2_2_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/34.895972"},{"key":"e_1_2_2_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.114"},{"key":"e_1_2_2_47_1","doi-asserted-by":"crossref","unstructured":"Szegedy C. Liu W. Jia Y. Sermanet P. Reed S. Anguelov D. Erhan D. Vanhoucke V. and Rabinovich A. 2014. Going deeper with convolutions. arXiv preprint arXiv:1409.4842.","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"e_1_2_2_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.220"},{"key":"e_1_2_2_49_1","first-page":"3","article-title":"Visualizing high-dimensional data using t-sne","volume":"9","author":"van der Maaten L.","year":"2008","unstructured":"van der Maaten, L., and Hinton, G. 2008. Visualizing high-dimensional data using t-sne. Journal of Machine Learning Research 9, 3 (Nov.), 2579--2605.","journal-title":"Journal of Machine Learning Research"},{"key":"e_1_2_2_50_1","doi-asserted-by":"publisher","unstructured":"Wang J. Song Y. Leung T. Rosenberg C. Wang J. Philbin J. Chen B. and Wu Y. 2014. Learning fine-grained image similarity with deep ranking. CoRR abs\/1404.4661. 10.1109\/CVPR.2014.180","DOI":"10.1109\/CVPR.2014.180"},{"key":"e_1_2_2_51_1","volume-title":"The IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","author":"Wang F.","unstructured":"Wang, F., Kang, L., and Li, Y. 2015. Sketch-based 3d shape retrieval using convolutional neural networks. In The IEEE Conference on Computer Vision and Pattern Recognition (CVPR)."},{"key":"e_1_2_2_52_1","doi-asserted-by":"publisher","unstructured":"Xiao J. Ehinger K. A. Hays J. Torralba A. and Oliva A. 2014. Sun database: Exploring a large collection of scene categories. International Journal of Computer Vision 1--20. 10.1007\/s11263-014-0748-y","DOI":"10.1007\/s11263-014-0748-y"},{"key":"e_1_2_2_53_1","volume-title":"British Machine Vision Conference (BMVC).","author":"Yu Q.","unstructured":"Yu, Q., Yang, Y., Song, Y.-Z., Xiang, T., and Hospedales, T. 2015. Sketch-a-net that beats humans. In British Machine Vision Conference (BMVC)."},{"key":"e_1_2_2_54_1","volume-title":"The IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","author":"Yu Q.","unstructured":"Yu, Q., Liu, F., Song, Y., Xiang, T., Hospedales, T., and Loy, C. C. 2016. Sketch me that shoe. In The IEEE Conference on Computer Vision and Pattern Recognition (CVPR)."},{"key":"e_1_2_2_55_1","doi-asserted-by":"crossref","unstructured":"Zeiler M. D. and Fergus R. 2014. Visualizing and understanding convolutional networks. In Computer Vision--ECCV 2014. Springer 818--833.","DOI":"10.1007\/978-3-319-10590-1_53"},{"key":"e_1_2_2_56_1","volume-title":"The IEEE Conference on Computer Vision and Pattern Recognition (CVPR).","author":"Zhou T.","unstructured":"Zhou, T., Jae Lee, Y., Yu, S. X., and Efros, A. A. 2015. Flowweb: Joint image set alignment by weaving consistent, pixel-wise correspondences. In The IEEE Conference on Computer Vision and Pattern Recognition (CVPR)."},{"key":"e_1_2_2_57_1","doi-asserted-by":"publisher","DOI":"10.1145\/2601097.2601145"}],"container-title":["ACM Transactions on Graphics"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/2897824.2925954","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/2897824.2925954","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/2897824.2925954","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,18]],"date-time":"2025-11-18T09:28:09Z","timestamp":1763458089000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/2897824.2925954"}},"subtitle":["learning to retrieve badly drawn bunnies"],"short-title":[],"issued":{"date-parts":[[2016,7,11]]},"references-count":57,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2016,7,11]]}},"alternative-id":["10.1145\/2897824.2925954"],"URL":"https:\/\/doi.org\/10.1145\/2897824.2925954","relation":{},"ISSN":["0730-0301","1557-7368"],"issn-type":[{"value":"0730-0301","type":"print"},{"value":"1557-7368","type":"electronic"}],"subject":[],"published":{"date-parts":[[2016,7,11]]},"assertion":[{"value":"2016-07-11","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}