{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,3]],"date-time":"2026-06-03T07:26:05Z","timestamp":1780471565488,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":49,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,6,8]],"date-time":"2022-06-08T00:00:00Z","timestamp":1654646400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,6,8]]},"DOI":"10.1145\/3517031.3529628","type":"proceedings-article","created":{"date-parts":[[2022,5,27]],"date-time":"2022-05-27T04:18:15Z","timestamp":1653625095000},"page":"1-7","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":9,"title":["Can Gaze Inform Egocentric Action Recognition?"],"prefix":"10.1145","author":[{"given":"Zehua","family":"Zhang","sequence":"first","affiliation":[{"name":"Computer Science, Indiana University Bloomington, United States"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"David","family":"Crandall","sequence":"additional","affiliation":[{"name":"Computer Science, Indiana University, United States"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Michael","family":"Proulx","sequence":"additional","affiliation":[{"name":"Meta Reality Lab, Meta Platforms, United States"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sachin","family":"Talathi","sequence":"additional","affiliation":[{"name":"Meta Reality Lab, Meta Platforms, United States"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Abhishek","family":"Sharma","sequence":"additional","affiliation":[{"name":"Meta Reality Lab, Meta Platforms, United States"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2022,6,8]]},"reference":[{"key":"e_1_3_2_3_1_1","unstructured":"Gedas Bertasius Heng Wang and Lorenzo Torresani. 2021. Is Space-Time Attention All You Need for Video Understanding?arXiv preprint arXiv:2102.05095(2021)."},{"key":"e_1_3_2_3_2_1","doi-asserted-by":"crossref","unstructured":"Anubhav Bhatti Behnam Behinaein Dirk Rodenburg Paul Hungler and Ali Etemad. 2021. Attentive Cross-modal Connections for Deep Multimodal Wearable-based Emotion Recognition. CoRR abs\/2108.02241(2021).","DOI":"10.1109\/ACIIW52867.2021.9666360"},{"key":"e_1_3_2_3_3_1","doi-asserted-by":"publisher","DOI":"10.1167\/14.3.29"},{"key":"e_1_3_2_3_4_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"e_1_3_2_3_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.502"},{"key":"e_1_3_2_3_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.691"},{"key":"e_1_3_2_3_7_1","doi-asserted-by":"publisher","DOI":"10.1007\/s00221-018-5369-1"},{"key":"e_1_3_2_3_8_1","volume-title":"Multixnet: Multiclass multistage multimodal motion prediction. arXiv preprint arXiv:2006.02000(2020).","author":"Djuric Nemanja","year":"2020","unstructured":"Nemanja Djuric, Henggang Cui, Zhaoen Su, Shangxuan Wu, Huahua Wang, Fang-Chieh Chou, Luisa\u00a0San Martin, Song Feng, Rui Hu, Yang Xu, 2020. Multixnet: Multiclass multistage multimodal motion prediction. arXiv preprint arXiv:2006.02000(2020)."},{"key":"e_1_3_2_3_9_1","unstructured":"Alexey Dosovitskiy Lucas Beyer Alexander Kolesnikov Dirk Weissenborn Xiaohua Zhai Thomas Unterthiner Mostafa Dehghani Matthias Minderer Georg Heigold Sylvain Gelly 2020. An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929(2020)."},{"key":"e_1_3_2_3_10_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-33718-5_23"},{"key":"e_1_3_2_3_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.213"},{"key":"e_1_3_2_3_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2020.2992889"},{"key":"e_1_3_2_3_13_1","doi-asserted-by":"crossref","unstructured":"Mozhdeh Gheini Xiang Ren and Jonathan May. 2021. Cross-Attention is All You Need: Adapting Pretrained Transformers for Machine Translation. arxiv:2104.08771\u00a0[cs.CL]","DOI":"10.18653\/v1\/2021.emnlp-main.132"},{"key":"e_1_3_2_3_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3352763"},{"key":"e_1_3_2_3_15_1","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pone.0064937"},{"key":"e_1_3_2_3_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/TIP.2020.3007841"},{"key":"e_1_3_2_3_17_1","volume-title":"The Grace Hopper Celebration of Women in Computing, Vol.\u00a04.","author":"Iqbal T","year":"2004","unstructured":"Shamsi\u00a0T Iqbal and Brian\u00a0P Bailey. 2004. Using eye gaze patterns to identify user tasks. In The Grace Hopper Celebration of Women in Computing, Vol.\u00a04. 2004."},{"key":"e_1_3_2_3_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW.2019.00540"},{"key":"e_1_3_2_3_19_1","unstructured":"Will Kay Joao Carreira Karen Simonyan Brian Zhang Chloe Hillier Sudheendra Vijayanarasimhan Fabio Viola Tim Green Trevor Back Paul Natsev 2017. The kinetics human action video dataset. arXiv preprint arXiv:1705.06950(2017)."},{"key":"e_1_3_2_3_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00559"},{"key":"e_1_3_2_3_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2011.5995406"},{"key":"e_1_3_2_3_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/IROS.2018.8594049"},{"key":"e_1_3_2_3_23_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.preteyeres.2006.01.002"},{"key":"e_1_3_2_3_24_1","volume-title":"Beholder: Gaze and Actions in First Person Video. arxiv:2006.00626\u00a0[cs.CV]","author":"Li Yin","year":"2020","unstructured":"Yin Li, Miao Liu, and James\u00a0M. Rehg. 2020. In the Eye of the Beholder: Gaze and Actions in First Person Video. arxiv:2006.00626\u00a0[cs.CV]"},{"key":"e_1_3_2_3_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00752"},{"key":"e_1_3_2_3_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01157"},{"key":"e_1_3_2_3_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW.2019.00543"},{"key":"e_1_3_2_3_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00376"},{"key":"e_1_3_2_3_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.209"},{"key":"e_1_3_2_3_30_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.cub.2018.03.008"},{"key":"e_1_3_2_3_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV48630.2021.00111"},{"key":"e_1_3_2_3_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW50498.2020.00054"},{"key":"e_1_3_2_3_33_1","doi-asserted-by":"publisher","DOI":"10.1146\/annurev-vision-082114-035733"},{"key":"e_1_3_2_3_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298691"},{"key":"e_1_3_2_3_35_1","unstructured":"Karen Simonyan and Andrew Zisserman. 2014. Two-stream convolutional networks for action recognition in videos. arXiv preprint arXiv:1406.2199(2014)."},{"key":"e_1_3_2_3_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.287"},{"key":"e_1_3_2_3_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW.2009.5204354"},{"key":"e_1_3_2_3_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.01019"},{"key":"e_1_3_2_3_39_1","doi-asserted-by":"crossref","unstructured":"Swathikiran Sudhakaran and Oswald Lanz. 2018. Attention is all we need: Nailing down object-centric attention for egocentric activity recognition. arXiv preprint arXiv:1807.11794(2018).","DOI":"10.1109\/CVPR.2019.01019"},{"key":"e_1_3_2_3_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"e_1_3_2_3_41_1","unstructured":"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit Llion Jones Aidan\u00a0N Gomez Lukasz Kaiser and Illia Polosukhin. 2017. Attention is all you need. arXiv preprint arXiv:1706.03762(2017)."},{"key":"e_1_3_2_3_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00813"},{"key":"e_1_3_2_3_43_1","volume-title":"Eye movements and vision","author":"Yarbus L","unstructured":"Alfred\u00a0L Yarbus. 2013. Eye movements and vision. Springer."},{"key":"e_1_3_2_3_44_1","unstructured":"Linwei Ye Mrigank Rochan Zhi Liu and Yang Wang. 2019. Cross-Modal Self-Attention Network for Referring Image Segmentation. CoRR abs\/1904.04745(2019)."},{"key":"e_1_3_2_3_45_1","volume-title":"British Machine Vision Conference (BMVC).","author":"Zhang Zehua","year":"2018","unstructured":"Zehua Zhang, Sven Bambach, Chen Yu, and David\u00a0J Crandall. 2018. From Coarse Attention to Fine-Grained Gaze: A Two-stage 3D Fully Convolutional Network for Predicting Eye Gaze in First Person Video. In British Machine Vision Conference (BMVC)."},{"key":"e_1_3_2_3_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01136"},{"key":"e_1_3_2_3_47_1","volume-title":"Interaction Graphs for Object Importance Estimation in On-road Driving Videos. In 2020 IEEE International Conference on Robotics and Automation (ICRA). IEEE, 8920\u20138927","author":"Zhang Zehua","year":"2020","unstructured":"Zehua Zhang, Ashish Tawari, Sujitha Martin, and David Crandall. 2020b. Interaction Graphs for Object Importance Estimation in On-road Driving Videos. In 2020 IEEE International Conference on Robotics and Automation (ICRA). IEEE, 8920\u20138927."},{"key":"e_1_3_2_3_48_1","unstructured":"Zehua Zhang Chen Yu and David Crandall. 2019. A Self Validation Network for Object-Level Human Attention Estimation. In Advances in Neural Information Processing Systems. 14702\u201314713."},{"key":"e_1_3_2_3_49_1","doi-asserted-by":"crossref","unstructured":"B. Zhou A. Khosla Lapedriza. A. A. Oliva and A. Torralba. 2016. Learning Deep Features for Discriminative Localization.CVPR (2016).","DOI":"10.1109\/CVPR.2016.319"}],"event":{"name":"ETRA '22: 2022 Symposium on Eye Tracking Research and Applications","location":"Seattle WA USA","acronym":"ETRA '22","sponsor":["SIGGRAPH ACM Special Interest Group on Computer Graphics and Interactive Techniques","SIGCHI ACM Special Interest Group on Computer-Human Interaction"]},"container-title":["2022 Symposium on Eye Tracking Research and Applications"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3517031.3529628","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3517031.3529628","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T18:09:00Z","timestamp":1750183740000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3517031.3529628"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,6,8]]},"references-count":49,"alternative-id":["10.1145\/3517031.3529628","10.1145\/3517031"],"URL":"https:\/\/doi.org\/10.1145\/3517031.3529628","relation":{},"subject":[],"published":{"date-parts":[[2022,6,8]]},"assertion":[{"value":"2022-06-08","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}