{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,19]],"date-time":"2026-02-19T02:08:45Z","timestamp":1771466925307,"version":"3.50.1"},"reference-count":47,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2013,9,14]],"date-time":"2013-09-14T00:00:00Z","timestamp":1379116800000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2014,2]]},"DOI":"10.1007\/s11263-013-0655-7","type":"journal-article","created":{"date-parts":[[2013,9,13]],"date-time":"2013-09-13T04:01:00Z","timestamp":1379044860000},"page":"282-296","source":"Crossref","is-referenced-by-count":98,"title":["Detecting People Looking at Each Other in Videos"],"prefix":"10.1007","volume":"106","author":[{"given":"M. J.","family":"Marin-Jimenez","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"A.","family":"Zisserman","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"M.","family":"Eichner","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"V.","family":"Ferrari","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2013,9,14]]},"reference":[{"key":"655_CR1","doi-asserted-by":"crossref","unstructured":"Andriluka, M., Roth, S., & Schiele, B. (2009). Pictorial structures revisited: People detection and articulated pose estimation. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition.","DOI":"10.1109\/CVPR.2009.5206754"},{"key":"655_CR2","doi-asserted-by":"crossref","unstructured":"Ba, S., & Odobez, J. M. (2005). Evaluation of multiple cue head pose estimation algorithms in natural environements. In Proceedings of the IEEE International Conference on Multimedia and Expo.","DOI":"10.1109\/ICME.2005.1521675"},{"issue":"1","key":"655_CR3","doi-asserted-by":"crossref","first-page":"16","DOI":"10.1109\/TSMCB.2008.927274","volume":"39","author":"S Ba","year":"2009","unstructured":"Ba, S., & Odobez, J. M. (2009). Recognizing visual focus of attention from head pose in natural meetings. IEEE Transactions on Systems, Man, and Cybernetics, Part B: Cybernetics, 39(1), 16\u201333.","journal-title":"IEEE Transactions on Systems, Man, and Cybernetics, Part B: Cybernetics"},{"key":"655_CR4","doi-asserted-by":"crossref","unstructured":"Benfold, B., & Reid, I. (2008). Colour invariant head pose classification in low resolution video. In Proceedings of the British Machine Vision Conference.","DOI":"10.5244\/C.22.49"},{"key":"655_CR5","doi-asserted-by":"crossref","first-page":"1063","DOI":"10.1109\/TPAMI.2003.1227983","volume":"25","author":"V Blanz","year":"2003","unstructured":"Blanz, V., & Vetter, T. (2003). Face recognition based on fitting a 3d morphable model. IEEE Transactions on Pattern Analysis and Machine Intelligence, 25, 1063\u20131074.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"655_CR6","doi-asserted-by":"crossref","unstructured":"Bourdev, L., Maji, S., Brox, T., & Malik, J. (2010). Detecting people using mutually consistent poselet activations. In Proceedings of the European Conference on Computer Vision.","DOI":"10.1007\/978-3-642-15567-3_13"},{"key":"655_CR7","doi-asserted-by":"crossref","unstructured":"Cour, T., Sapp, B., Jordan, C., & Taskar, B. (2009). Learning from ambiguously labeled images. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition.","DOI":"10.1109\/CVPR.2009.5206667"},{"key":"655_CR8","doi-asserted-by":"crossref","unstructured":"Dalal, N., & Triggs, B. (2005). Histogram of Oriented Gradients for Human Detection. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, vol,2, (pp. 886\u2013893).","DOI":"10.1109\/CVPR.2005.177"},{"issue":"12","key":"655_CR9","doi-asserted-by":"crossref","first-page":"31","DOI":"10.1016\/S0004-3702(96)00034-3","volume":"89","author":"TG Dietterich","year":"1997","unstructured":"Dietterich, T. G., Lathrop, R. H., & Lozano-P \u00e9rez, T. (1997). Solving the multiple instance problem with axis-parallel rectangles. Artificial Intelligence, 89(12), 31\u201371.","journal-title":"Artificial Intelligence"},{"key":"655_CR10","doi-asserted-by":"crossref","unstructured":"Everingham, M., & Zisserman, A. (2005). Identifying individuals in video by combining generative and discriminative head models. In Proceedings of the International Conference on Computer Vision.","DOI":"10.1109\/ICCV.2005.116"},{"key":"655_CR11","unstructured":"Everingham, M., Sivic, J., & Zisserman, A. (2006). Hello! My name is... Buffy: Automatic naming of characters in TV video. In Proceedings of the British Machine Vision Conference."},{"issue":"2","key":"655_CR12","doi-asserted-by":"crossref","first-page":"303","DOI":"10.1007\/s11263-009-0275-4","volume":"88","author":"M Everingham","year":"2010","unstructured":"Everingham, M., Van Gool, L., Williams, C. K. I., Winn, J., & Zisserman, A. (2010). The pascal visual object classes (voc) challenge. International Journal of Computer Vision, 88(2), 303\u2013338.","journal-title":"International Journal of Computer Vision"},{"key":"655_CR13","doi-asserted-by":"crossref","unstructured":"Fathi, A., Hodgins, J., & Regh, J. (2012). Social interactions: A first-person perspective. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition.","DOI":"10.1109\/CVPR.2012.6247805"},{"issue":"9","key":"655_CR14","doi-asserted-by":"crossref","first-page":"1627","DOI":"10.1109\/TPAMI.2009.167","volume":"32","author":"P Felzenszwalb","year":"2010","unstructured":"Felzenszwalb, P., Girshick, R., McAllester, D., & Ramanan, D. (2010). Object detection with discriminatively trained part based models. IEEE Transactions on Pattern Analysis and Machine Intelligence, 32(9), 1627\u20131645.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"655_CR15","doi-asserted-by":"crossref","unstructured":"Ferrari, V., Tuytelaars, T., & Van Gool, L. (2001). Real-time affine region tracking and coplanar grouping. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition.","DOI":"10.1109\/CVPR.2001.990964"},{"key":"655_CR16","doi-asserted-by":"crossref","unstructured":"Ferrari, V., Marin, M., & Zisserman, A. (2008). Progressive search space reduction for human pose estimation. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition.","DOI":"10.1109\/CVPR.2008.4587468"},{"key":"655_CR17","doi-asserted-by":"crossref","unstructured":"Ferrari, V., Marin-Jimenez, M., & Zisserman, A. (2009). Pose search: Retrieving people using their pose. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition.","DOI":"10.1109\/CVPR.2009.5206495"},{"key":"655_CR18","unstructured":"Jones, M. & Viola, P. (2003). Fast multi-view face detection. Technical Report TR2003-96, MERL."},{"key":"655_CR19","unstructured":"Kim, W. H., & Kim, J. N. (2009). An adaptive shot change detection algorithm using an average of absolute difference histogram within extension sliding window. In IEEE International Symposium on Consumer Electronics."},{"key":"655_CR20","unstructured":"Kl\u00e4ser, A., Marsza\u0142ek, M., Schmid, C., & Zisserman, A. (2010). Human focused action localization in video (pp. 219\u2013233). ECCV-International Workshop on Sign, Gesture, Activity."},{"key":"655_CR21","doi-asserted-by":"crossref","unstructured":"Laptev, I., Marsza\u0142ek, M., Schmid, C., & Rozenfeld, B. (2008). Learning realistic human actions from movies. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition.","DOI":"10.1109\/CVPR.2008.4587756"},{"key":"655_CR22","doi-asserted-by":"crossref","unstructured":"Liu, J., Luo, J., & Shah, M. (2009). Recognizing realistic actions from videos. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition.","DOI":"10.1109\/CVPR.2009.5206744"},{"key":"655_CR23","unstructured":"Mar\u00edn-Jim\u00e9nez, M., Zisserman, A., & Ferrari, V. (2011). Here\u2019s looking at you kid. Detecting people looking at each other in videos. In Proceedings of the British Machine Vision Conference."},{"key":"655_CR24","unstructured":"Mar\u00edn-Jim\u00e9nez, M., P\u00e9rez de la Blanca, N., & Mendoza, M. (2012). Human action recognition from simple feature pooling. Pattern Analysis and Applications."},{"key":"655_CR25","doi-asserted-by":"crossref","first-page":"607","DOI":"10.1109\/TPAMI.2008.106","volume":"31","author":"E Murphy-Chutorian","year":"2009","unstructured":"Murphy-Chutorian, E., & Trivedi, M. M. (2009). Head pose estimation in computer vision: A survey. IEEE Transactions on Pattern Analysis and Machine Intelligence, 31, 607\u2013626.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"655_CR26","first-page":"1197","volume":"8","author":"M Osadchy","year":"2007","unstructured":"Osadchy, M., Cun, Y., & Miller, M. (2007). Synergistic face detection and pose estimation with energy-based models. Journal of Machine Learning Research, 8, 1197\u20131215.","journal-title":"Journal of Machine Learning Research"},{"key":"655_CR27","doi-asserted-by":"crossref","unstructured":"Park, S. & Aggarwal, J. (2004). A hierarchical bayesian network for event recognition of human actions and interactions. Association For Computing Machinery Multimedia Systems Journal.","DOI":"10.1007\/s00530-004-0148-1"},{"key":"655_CR28","unstructured":"Patron-Perez, A., Marszalek, M., Reid, I., & Zisserman, A. (2010). High Five: Recognising human interactions in TV shows. In Proceedings of the British Machine Vision Conference."},{"issue":"12","key":"655_CR29","doi-asserted-by":"crossref","first-page":"2441","DOI":"10.1109\/TPAMI.2012.24","volume":"34","author":"A Patron-Perez","year":"2012","unstructured":"Patron-Perez, A., Marszalek, M., Reid, I., & Zisserman, A. (2012). Structured learning of human interactions in tv shows. IEEE Transactions on Pattern Analysis and Machine Intelligence, 34(12), 2441\u20132453.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"655_CR30","doi-asserted-by":"crossref","unstructured":"Raptis, M., Kokkinos, I., & Soatto, S. (2012). Discovering discriminative action parts from mid-level video representations. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition.","DOI":"10.1109\/CVPR.2012.6247807"},{"key":"655_CR31","volume-title":"Gaussian processes for machine learning","author":"CE Rasmussen","year":"2006","unstructured":"Rasmussen, C. E., & Williams, C. K. I. (2006). Gaussian processes for machine learning. Cambridge, MA: MIT Press."},{"key":"655_CR32","doi-asserted-by":"crossref","unstructured":"Sadanand, S., & Corso, J. (2012). Action bank: A high-level representation of activity in video. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition.","DOI":"10.1109\/CVPR.2012.6247806"},{"key":"655_CR33","doi-asserted-by":"crossref","unstructured":"Sapp, B., Toshev, A., & Taskar, B. (2010). Cascaded models for articulated pose estimation. In Proceedings of the European Conference on Computer Vision.","DOI":"10.1007\/978-3-642-15552-9_30"},{"key":"655_CR34","unstructured":"Shi, J., & Tomasi, C. (1994). Good features to track (pp. 593\u2013600). In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition."},{"issue":"1","key":"655_CR35","first-page":"1615","volume":"25","author":"T Sim","year":"2003","unstructured":"Sim, T., Baker, S., & Bsat, M. (2003). The CMU pose, illumination, and expression database. IEEE Transactions on Pattern Analysis and Machine Intelligence, 25(1), 1615\u20131618.","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"655_CR36","doi-asserted-by":"crossref","unstructured":"Sivic, J., Everingham, M., & Zisserman, A. (2005). Person spotting: Video shot retrieval for face sets. In Proceedings of the ACM International Conference on Image and Video Retrieval.","DOI":"10.1007\/11526346_26"},{"key":"655_CR37","doi-asserted-by":"crossref","unstructured":"Sivic, J., Everingham, M., & Zisserman, A. (2009). \u201cWho are you?\u201d: Learning person specific classifiers from video. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition.","DOI":"10.1109\/CVPR.2009.5206513"},{"key":"655_CR38","doi-asserted-by":"crossref","unstructured":"Tang, S., Andriluka, M., & Schiele, B. (2012). Detection and tracking of occluded people. In Proceedings of the British Machine Vision Conference.","DOI":"10.5244\/C.26.9"},{"key":"655_CR39","unstructured":"Tu, Z. (2005). Probabilistic boosting-tree: Learning discriminative models for classification, recognition, and clustering. In Proceedings of the International Conference on Computer Vision."},{"key":"655_CR40","doi-asserted-by":"crossref","unstructured":"Waltisberg, W., Yao, A., Gall, J., Gool, LV. (2010). Variations of a Hough-voting action recognition system. In Proceedings of the International Conference on Pattern Recognition (ICPR) 2010 Contests.","DOI":"10.1007\/978-3-642-17711-8_31"},{"key":"655_CR41","unstructured":"Website. (2005). INRIA person dataset. http:\/\/pascal.inrialpes.fr\/data\/human\/ ."},{"key":"655_CR42","unstructured":"Website. (2010). Deformable parts model code. http:\/\/www.cs.brown.edu\/pff\/latent\/ ."},{"key":"655_CR43","unstructured":"Website. (2011a). GPML Matlab code. http:\/\/www.gaussianprocess.org\/gpml\/code\/matlab\/doc\/ ."},{"key":"655_CR44","unstructured":"Website. (2011b). LAEO annotations. http:\/\/www.robots.ox.ac.uk\/vgg\/data\/laeo\/ ."},{"key":"655_CR45","unstructured":"Website. (2011c). LAEO project. http:\/\/www.robots.ox.ac.uk\/vgg\/research\/laeo\/ ."},{"key":"655_CR46","doi-asserted-by":"crossref","unstructured":"Yang, Y., Baker, S., Kannan, A., & Ramanan, D. (2012). Recognizing proxemics in personal photos. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition.","DOI":"10.1109\/CVPR.2012.6248095"},{"key":"655_CR47","unstructured":"Zhu, X., & Ramanan, D. (2012). Face detection, pose estimation and landmark localization in the wild. In Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition."}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-013-0655-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/article\/10.1007\/s11263-013-0655-7\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"http:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-013-0655-7","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2019,7,24]],"date-time":"2019-07-24T02:09:10Z","timestamp":1563934150000},"score":1,"resource":{"primary":{"URL":"http:\/\/link.springer.com\/10.1007\/s11263-013-0655-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2013,9,14]]},"references-count":47,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2014,2]]}},"alternative-id":["655"],"URL":"https:\/\/doi.org\/10.1007\/s11263-013-0655-7","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2013,9,14]]}}}