{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,15]],"date-time":"2026-04-15T19:48:28Z","timestamp":1776282508241,"version":"3.50.1"},"reference-count":37,"publisher":"Springer Science and Business Media LLC","issue":"5","license":[{"start":{"date-parts":[[2019,11,11]],"date-time":"2019-11-11T00:00:00Z","timestamp":1573430400000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2019,11,11]],"date-time":"2019-11-11T00:00:00Z","timestamp":1573430400000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2020,5]]},"DOI":"10.1007\/s11263-019-01255-4","type":"journal-article","created":{"date-parts":[[2019,11,11]],"date-time":"2019-11-11T12:03:21Z","timestamp":1573473801000},"page":"1061-1075","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":14,"title":["Efficient Object Annotation via Speaking and Pointing"],"prefix":"10.1007","volume":"128","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5241-481X","authenticated-orcid":false,"given":"Michael","family":"Gygli","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Vittorio","family":"Ferrari","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2019,11,11]]},"reference":[{"key":"1255_CR1","unstructured":"Bearman, A., Russakovsky, O., Ferrari, V., Fei-Fei, L. (2016). What\u2019s the point: Semantic segmentation with point supervision. In: Proceedings of the European conference on computer vision."},{"key":"1255_CR2","unstructured":"Bolt, RA. (1980). \u201cPut-that-there\u201d: Voice and gesture at the graphics interface. In: SIGGRAPH."},{"key":"1255_CR3","unstructured":"Clarkson, E., Clawson, J., Lyons, K., Starner, T. (2005). An empirical study of typing rates on mini-qwerty keyboards. In: CHI."},{"key":"1255_CR4","unstructured":"Dai, D. (2016). Towards cost-effective and performance-aware vision algorithms. Ph.D. thesis, ETH Zurich."},{"key":"1255_CR5","unstructured":"Damen, D., Doughty, H., Maria\u00a0Farinella, G., Fidler, S., Furnari, A., Kazakos E, et\u00a0al. (2018). Scaling egocentric vision: The epic-kitchens dataset. In: Proceedings of the European conference on computer vision."},{"key":"1255_CR6","unstructured":"Deng, J., Dong, W., Socher, R., Li, LJ., Li, K., Fei-fei, L. (2009). Imagenet: A large-scale hierarchical image database. In: IEEE conference on computer vision and pattern recognition."},{"key":"1255_CR7","unstructured":"Deng, J., Russakovsky, O., Krause, J., Bernstein, MS., Berg, A., Fei-Fei L (2014). Scalable multi-label annotation. In: CHI."},{"key":"1255_CR8","doi-asserted-by":"publisher","first-page":"945","DOI":"10.1080\/13506280902834720","volume":"17","author":"KA Ehinger","year":"2009","unstructured":"Ehinger, K. A., Hidalgo-Sotelo, B., Torralba, A., & Oliva, A. (2009). Modelling search for people in 900 scenes: A combined source model of eye guidance. Visual Cognition, 17, 945\u2013978.","journal-title":"Visual Cognition"},{"key":"1255_CR9","unstructured":"Gygli, M., Ferrari, V. (2019). Fast object class labelling via speech. In: Proceedings of the IEEE conference on computer vision and pattern recognition."},{"key":"1255_CR10","unstructured":"Harwath, D., Recasens, A., Sur\u00eds, D., Chuang, G., Torralba, A., Glass J (2018). Jointly discovering visual objects and spoken words from raw sensory input. In: Proceedings of the European conference on computer vision (ECCV)."},{"key":"1255_CR11","unstructured":"Hauptmann, AG. (1989). Speech and gestures for graphic image manipulation. In: ACM SIGCHI."},{"key":"1255_CR12","volume-title":"Attention and effort","author":"D Kahneman","year":"1973","unstructured":"Kahneman, D. (1973). Attention and effort. Englewood Cliffs: Prentice-Hall."},{"key":"1255_CR13","unstructured":"Karat, CM., Halverson, C., Horn, D., Karat, J. (1999). Patterns of entry and correction in large vocabulary continuous speech recognition systems. In: ACM SIGCHI, ACM."},{"key":"1255_CR14","unstructured":"Krishna, RA., Hata, K., Chen, S., Kravitz, J., Shamma, DA., Fei-Fei, L., Bernstein, MS. (2016). Embracing error to enable rapid crowdsourcing. In: CHI."},{"key":"1255_CR15","unstructured":"Kuznetsova, A., Rom, H., Alldrin, N., Uijlings, J., Krasin, I., Pont-Tuset, J., Kamali, S., Popov, S., Malloci, M., Duerig, T., Ferrari, V. (2018). The open images dataset V4: Unified image classification, object detection, and visual relationship detection at scale. arXiv preprint arXiv:1811.00982."},{"key":"1255_CR16","unstructured":"Laradji, IH., Rostamzadeh, N., Pinheiro, PO., Vazquez, D., Schmidt, M. (2018). Where are the blobs: Counting by localization with point supervision. arXiv preprint arXiv:1807.09856."},{"key":"1255_CR17","unstructured":"Lin, TY., Maire, M., Belongie, S., Hays, J., Perona, P., Ramanan, D., Doll\u00e1r P, Zitnick, C. (2014). Microsoft COCO: Common objects in context. In: European conference on computer vision."},{"key":"1255_CR18","doi-asserted-by":"publisher","first-page":"684","DOI":"10.1111\/j.1467-9280.2005.01596.x","volume":"16","author":"A Lleras","year":"2005","unstructured":"Lleras, A., Rensink, R. A., & Enns, J. T. (2005). Rapid resumption of interrupted visual search: New insights on the interaction between vision and memory. Psychological Science, 16, 684\u2013688.","journal-title":"Psychological Science"},{"key":"1255_CR19","unstructured":"Manen, S., Gygli, M., Dai, D., Van\u00a0Gool, L. (2017). Pathtrack: Fast trajectory annotation with path supervision. In: IEEE international conference on computer vision."},{"key":"1255_CR20","unstructured":"Mettes, P., van Gemert, J.C., Snoek, C.G. (2016). Spot on: Action localization from pointly-supervised proposals. In: European conference on computer vision."},{"key":"1255_CR21","unstructured":"Mikolov, T., Chen, K., Corrado, G., Dean, J. (2013). Efficient estimation of word representations in vector space. arXiv preprint arXiv:1301.3781."},{"key":"1255_CR22","doi-asserted-by":"publisher","first-page":"443","DOI":"10.1016\/0022-2836(70)90057-4","volume":"48","author":"SB Needleman","year":"1970","unstructured":"Needleman, S. B., & Wunsch, C. D. (1970). A general method applicable to the search for similarities in the amino acid sequence of two proteins. Journal of Molecular Biology, 48, 443\u2013453.","journal-title":"Journal of Molecular Biology"},{"key":"1255_CR23","unstructured":"Oviatt, S. (1996). Multimodal interfaces for dynamic interactive maps. In: ACM SIGCHI."},{"key":"1255_CR24","volume-title":"The human-computer interaction handbook: Fundamentals, evolving technologies and emerging applications","author":"S Oviatt","year":"2003","unstructured":"Oviatt, S. (2003). Multimodal interfaces. In J. A. Jacko & A. Sears (Eds.), The human-computer interaction handbook: Fundamentals, evolving technologies and emerging applications. Boca Raton: CRC Press."},{"key":"1255_CR25","unstructured":"Oviatt, S., DeAngeli, A., Kuhn, K. (1997). Integration and synchronization of input modes during multimodal human\u2013computer interaction. In: CHI."},{"key":"1255_CR26","doi-asserted-by":"crossref","unstructured":"Papadopoulos, DP., Uijlings, JR., Keller, F., Ferrari, V. (2017a). Extreme clicking for efficient object annotation. In: Proceedings of the IEEE international conference on computer vision.","DOI":"10.1109\/ICCV.2017.528"},{"key":"1255_CR27","doi-asserted-by":"crossref","unstructured":"Papadopoulos, DP., Uijlings, JR., Keller, F., Ferrari, V. (2017b). Training object class detectors with click supervision. In: CVPR.","DOI":"10.1109\/CVPR.2017.27"},{"key":"1255_CR28","unstructured":"Pausch, R., Leatherby, JH. (1991). An empirical study: Adding voice input to a graphical editor. In: Journal of the American voice input\/output society."},{"key":"1255_CR29","unstructured":"Pont-Tuset, J., Gygli, M., Ferrari, V. (2019). Natural vocabulary emerges from free-form annotations. arXiv preprint arXiv:1906.01542."},{"key":"1255_CR30","doi-asserted-by":"publisher","first-page":"1457","DOI":"10.1080\/17470210902816461","volume":"62","author":"K Rayner","year":"2009","unstructured":"Rayner, K. (2009). Eye movements and attention in reading, scene perception, and visual search. Quarterly Journal of Experimental Psychology, 62, 1457\u20131506.","journal-title":"Quarterly Journal of Experimental Psychology"},{"key":"1255_CR31","doi-asserted-by":"publisher","first-page":"211","DOI":"10.1007\/s11263-015-0816-y","volume":"115","author":"O Russakovsky","year":"2015","unstructured":"Russakovsky, O., Deng, J., Su, H., Krause, J., Satheesh, S., Ma, S., et al. (2015a). Imagenet large scale visual recognition challenge. International Journal of Computer Vision, 115, 211\u2013252.","journal-title":"International Journal of Computer Vision"},{"key":"1255_CR32","doi-asserted-by":"crossref","unstructured":"Russakovsky, O., Li, LJ., Fei-Fei, L. (2015b). Best of both worlds: Human\u2013machine collaboration for object annotation. In: Proceedings of the IEEE conference on computer vision and pattern recognition.","DOI":"10.1109\/CVPR.2015.7298824"},{"key":"1255_CR33","unstructured":"Su, H., Deng, J., Fei-Fei, L. (2012). Crowdsourcing annotations for visual object detection. In: AAAI human computation workshop."},{"key":"1255_CR34","unstructured":"Sun, C., Shrivastava, A., Singh, S., Gupta, A. (2017). Revisiting unreasonable effectiveness of data in deep learning era. In: Proceedings of the IEEE international conference on computer vision."},{"key":"1255_CR35","unstructured":"Vaidyanathan, P., Prud, E., Pelz, JB., Alm, CO. (2018). SNAG : Spoken narratives and gaze dataset. In: Proceedings of Association for computational linguistics."},{"key":"1255_CR36","unstructured":"Vasudevan, AB., Dai, D., Van Gool, L. (2017). Object referring in visual scene with spoken language. In: Conference on computer vision and pattern recognition."},{"key":"1255_CR37","doi-asserted-by":"publisher","first-page":"852","DOI":"10.3758\/BF03194111","volume":"14","author":"DG Watson","year":"2007","unstructured":"Watson, D. G., & Inglis, M. (2007). Eye movements and time-based selection: Where do the eyes go in preview search? Psychonomic Bulletin & Review, 14, 852\u2013857.","journal-title":"Psychonomic Bulletin & Review"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-019-01255-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11263-019-01255-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-019-01255-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2020,11,10]],"date-time":"2020-11-10T00:17:12Z","timestamp":1604967432000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11263-019-01255-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,11,11]]},"references-count":37,"journal-issue":{"issue":"5","published-print":{"date-parts":[[2020,5]]}},"alternative-id":["1255"],"URL":"https:\/\/doi.org\/10.1007\/s11263-019-01255-4","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2019,11,11]]},"assertion":[{"value":"17 May 2019","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"16 October 2019","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 November 2019","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}