{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,20]],"date-time":"2026-06-20T21:23:48Z","timestamp":1781990628139,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":69,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,4,19]],"date-time":"2023-04-19T00:00:00Z","timestamp":1681862400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,4,19]]},"DOI":"10.1145\/3544548.3581006","type":"proceedings-article","created":{"date-parts":[[2023,4,20]],"date-time":"2023-04-20T04:28:44Z","timestamp":1681964924000},"page":"1-17","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":11,"title":["Identifying Multimodal Context Awareness Requirements for Supporting User Interaction with Procedural Videos"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9993-2718","authenticated-orcid":false,"given":"Georgianna","family":"Lin","sequence":"first","affiliation":[{"name":"Computer Science, University of Toronto, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4100-8327","authenticated-orcid":false,"given":"Jin Yi","family":"Li","sequence":"additional","affiliation":[{"name":"University of Toronto, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4479-4901","authenticated-orcid":false,"given":"Afsaneh","family":"Fazly","sequence":"additional","affiliation":[{"name":"Samsung Toronto AI Centre, Samsung Research America, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3979-1236","authenticated-orcid":false,"given":"Vladimir","family":"Pavlovic","sequence":"additional","affiliation":[{"name":"Rutgers University, United States"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0774-5964","authenticated-orcid":false,"given":"Khai","family":"Truong","sequence":"additional","affiliation":[{"name":"Computer Science, University of Toronto, Canada"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2023,4,19]]},"reference":[{"key":"e_1_3_3_2_1_1","doi-asserted-by":"publisher","DOI":"10.1145\/3301275.3302292"},{"key":"e_1_3_3_2_2_1","doi-asserted-by":"publisher","DOI":"10.1145\/3290605.3300562"},{"key":"e_1_3_3_2_3_1","doi-asserted-by":"crossref","unstructured":"Flora Amato Vincenzo Moscato Antonion Picariello Francesco Colace Massimo De\u00a0Santo Fabio\u00a0A. Schreiber and Letizia Tanca. 2017. Big Data Meets Digital Cultural Heritage: Design and Implementation of SCRABS A Smart Context-awaRe Browsing Assistant for Cultural EnvironmentS. Journal on Computing and Cultural Heritage(2017) 1\u201323.","DOI":"10.1145\/3012286"},{"key":"e_1_3_3_2_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/3123266.3126490"},{"key":"e_1_3_3_2_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/2047196.2047213"},{"key":"e_1_3_3_2_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/1357054.1357096"},{"key":"e_1_3_3_2_7_1","doi-asserted-by":"publisher","DOI":"10.2196\/10318"},{"key":"e_1_3_3_2_8_1","doi-asserted-by":"publisher","DOI":"10.1145\/3180155.3180238"},{"key":"e_1_3_3_2_9_1","volume-title":"Using thematic analysis in psychology. Qualitative research in psychology 3, 2","author":"Braun Virginia","year":"2006","unstructured":"Virginia Braun and Victoria Clarke. 2006. Using thematic analysis in psychology. Qualitative research in psychology 3, 2 (2006), 77\u2013101."},{"key":"e_1_3_3_2_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3173574.3173580"},{"key":"e_1_3_3_2_11_1","volume-title":"Proceedings of the 12th Language Resources and Evaluation Conference. 4352\u20134358","author":"Castro Santiago","year":"2020","unstructured":"Santiago Castro, Mahmoud Azab, Jonathan Stroud, Cristina Noujaim, Ruoyao Wang, Jia Deng, and Rada Mihalcea. 2020. LifeQA: A real-life dataset for video question answering. In Proceedings of the 12th Language Resources and Evaluation Conference. 4352\u20134358."},{"key":"e_1_3_3_2_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/ROMAN.2017.8172315"},{"key":"e_1_3_3_2_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/3411764.3445131"},{"key":"e_1_3_3_2_14_1","doi-asserted-by":"publisher","DOI":"10.1145\/3290605.3300931"},{"key":"e_1_3_3_2_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3449101"},{"key":"e_1_3_3_2_16_1","doi-asserted-by":"publisher","DOI":"10.1145\/2935334.2935386"},{"key":"e_1_3_3_2_18_1","doi-asserted-by":"publisher","DOI":"10.1145\/2401836.2401848"},{"key":"e_1_3_3_2_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/1978942.1979205"},{"key":"e_1_3_3_2_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/2642918.2647389"},{"key":"e_1_3_3_2_21_1","volume-title":"Towards Understanding People\u2019s Experiences of AI Computer Vision Fitness Instructor Apps. In Designing Interactive Systems Conference","author":"Garbett Andrew","year":"2021","unstructured":"Andrew Garbett, Ziedune Degutyte, James Hodge, and Arlene Astell. 2021. Towards Understanding People\u2019s Experiences of AI Computer Vision Fitness Instructor Apps. In Designing Interactive Systems Conference 2021. 1619\u20131637."},{"key":"e_1_3_3_2_22_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6713"},{"key":"e_1_3_3_2_23_1","doi-asserted-by":"crossref","unstructured":"Agust\u00edn Gravano Julia Hirschberg and \u0160tefan Be\u0148u\u0161. 2012. Affirmative Cue Words in Task-Oriented Dialogue. Computational Linguistics(2012) 1\u201339.","DOI":"10.1162\/COLI_a_00083"},{"key":"e_1_3_3_2_24_1","doi-asserted-by":"publisher","DOI":"10.1145\/2470654.2466235"},{"key":"e_1_3_3_2_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/2207676.2207766"},{"key":"e_1_3_3_2_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/2470654.2466149"},{"key":"e_1_3_3_2_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/2642918.2647366"},{"key":"e_1_3_3_2_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/2642918.2647400"},{"key":"e_1_3_3_2_29_1","doi-asserted-by":"publisher","DOI":"10.1145\/2807442.2807502"},{"key":"e_1_3_3_2_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/2638728.2641338"},{"key":"e_1_3_3_2_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3357251.3357581"},{"key":"e_1_3_3_2_32_1","doi-asserted-by":"publisher","unstructured":"Juho Kim. 2013. Toolscape: enhancing the learning experience of how-to videos. In Extended Abstracts on Human Factors in Computing Systems(CHI \u201913). 2707\u20132712. https:\/\/doi.org\/10.1145\/2468356.2479497","DOI":"10.1145\/2468356.2479497"},{"key":"e_1_3_3_2_33_1","volume-title":"Videodoc: Combining videos and lecture notes for a better learning experience. Master\u2019s thesis","author":"Krosnick P.","year":"2015","unstructured":"Rebecca\u00a0P. Krosnick. 2015. Videodoc: Combining videos and lecture notes for a better learning experience. Master\u2019s thesis. Massachusetts Institute of Technology, Cambridge, MA, United States."},{"key":"e_1_3_3_2_34_1","volume-title":"Proceedings of the 20th International Society for Music Information Retrieval Conference(ISMIR \u201919)","author":"Behrooz Morteza","year":"2019","unstructured":"Morteza Behrooz, Sarah Mennicken, Jennifer Thom,\u00a0Rohit Kumar, and Henriette Cramer. 2019. Augmenting Music Listening Experiences on Voice Assistants. In Proceedings of the 20th International Society for Music Information Retrieval Conference(ISMIR \u201919). 303\u2013310."},{"key":"e_1_3_3_2_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/3173574.3173859"},{"key":"e_1_3_3_2_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00171"},{"key":"e_1_3_3_2_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/3313831.3376479"},{"key":"e_1_3_3_2_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00999"},{"key":"e_1_3_3_2_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/3462244.3479902"},{"key":"e_1_3_3_2_40_1","doi-asserted-by":"crossref","unstructured":"Jie Lei Licheng Yu Tamara\u00a0L. Berg and Mohit Bansal. 2019. Tvqa+: Spatio-temporal grounding for video question answering. arXiv preprint arXiv:1904.11574(2019).","DOI":"10.18653\/v1\/2020.acl-main.730"},{"key":"e_1_3_3_2_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3441852.3471215"},{"key":"e_1_3_3_2_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/3446382.3448729"},{"key":"e_1_3_3_2_43_1","doi-asserted-by":"publisher","DOI":"10.1145\/2470654.2481301"},{"key":"e_1_3_3_2_44_1","doi-asserted-by":"publisher","DOI":"10.1145\/2858036.2858288"},{"key":"e_1_3_3_2_45_1","doi-asserted-by":"publisher","DOI":"10.1006\/ijhc.2001.0503"},{"key":"e_1_3_3_2_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.312"},{"key":"e_1_3_3_2_47_1","doi-asserted-by":"publisher","DOI":"10.1145\/2702123.2702209"},{"key":"e_1_3_3_2_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCIT.2007.211"},{"key":"e_1_3_3_2_49_1","doi-asserted-by":"publisher","DOI":"10.1145\/3290605.3300270"},{"key":"e_1_3_3_2_50_1","doi-asserted-by":"publisher","DOI":"10.1145\/3405755.3406119"},{"key":"e_1_3_3_2_51_1","doi-asserted-by":"publisher","DOI":"10.1145\/3296699"},{"key":"e_1_3_3_2_52_1","doi-asserted-by":"publisher","DOI":"10.1145\/3173574.3174214"},{"key":"e_1_3_3_2_53_1","doi-asserted-by":"publisher","DOI":"10.1145\/2540930.2540952"},{"key":"e_1_3_3_2_54_1","unstructured":"Chareen Snelson and Patt\u00a0R. Elison-Bowers. 2009. Using YouTube Videos to Engage the Affective Domain in E-Learning. (2009)."},{"key":"e_1_3_3_2_55_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.435"},{"key":"e_1_3_3_2_56_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.501"},{"key":"e_1_3_3_2_57_1","doi-asserted-by":"publisher","DOI":"10.1145\/2501988.2502038"},{"key":"e_1_3_3_2_58_1","doi-asserted-by":"publisher","DOI":"10.3390\/s20082376"},{"key":"e_1_3_3_2_59_1","volume-title":"Investigating ceramics, cuisine and culture: Past, present and future. Ceramics, Cuisine and Culture: The Archaeology and Science of Kitchen Pottery in the Ancient Mediterranean World","author":"Villing Alexandra","year":"2015","unstructured":"Alexandra Villing and Michela Spataro. 2015. Investigating ceramics, cuisine and culture: Past, present and future. Ceramics, Cuisine and Culture: The Archaeology and Science of Kitchen Pottery in the Ancient Mediterranean World, Oxbow Books, Oxford (2015), 1\u201325."},{"key":"e_1_3_3_2_60_1","doi-asserted-by":"publisher","DOI":"10.1145\/3173574.3173782"},{"key":"e_1_3_3_2_61_1","volume-title":"Exploring Interactions with Voice-Controlled TV. Computing Research Repository (CoRR) (May","author":"McRoberts Sarah","year":"2019","unstructured":"Sarah McRoberts, Joshua Wissbroecker,\u00a0Ruotong Wang, and F.\u00a0Maxwell Harper. 2019. Exploring Interactions with Voice-Controlled TV. Computing Research Repository (CoRR) (May 2019). http:\/\/arxiv.org\/abs\/1905.05851"},{"key":"e_1_3_3_2_62_1","doi-asserted-by":"crossref","unstructured":"Daniel\u00a0J. Weintraub Richard\u00a0F. Haines and Robert\u00a0J. Randle. 1985. Head-up display (HUD) utility. II-Runway to HUD transitions monitoring eye focus and decision times. (1985).","DOI":"10.1177\/154193128502900621"},{"key":"e_1_3_3_2_63_1","doi-asserted-by":"publisher","DOI":"10.1145\/3472749.3474795"},{"key":"e_1_3_3_2_64_1","volume-title":"HERO: Hierarchical encoder for video+language omni-representation pre-training. Empirical Methods in Natural Language Processing (EMNLP) (Sept.","author":"Li Linjie","year":"2020","unstructured":"Linjie Li, Yen-Chun Chen, Yu Cheng, Zhe Gan,\u00a0Licheng Yu, and Jingjing Liu. 2020. HERO: Hierarchical encoder for video+language omni-representation pre-training. Empirical Methods in Natural Language Processing (EMNLP) (Sept. 2020). https:\/\/arxiv.org\/abs\/2005.00200"},{"key":"e_1_3_3_2_65_1","doi-asserted-by":"publisher","DOI":"10.1145\/3041021.3054166"},{"key":"e_1_3_3_2_66_1","doi-asserted-by":"publisher","DOI":"10.1145\/2556288.2557368"},{"key":"e_1_3_3_2_67_1","doi-asserted-by":"publisher","DOI":"10.1145\/3491102.3502036"},{"key":"e_1_3_3_2_68_1","doi-asserted-by":"publisher","DOI":"10.1145\/2702123.2702305"},{"key":"e_1_3_3_2_69_1","doi-asserted-by":"publisher","DOI":"10.1145\/1111449.1111479"},{"key":"e_1_3_3_2_70_1","doi-asserted-by":"publisher","DOI":"10.1145\/3232077"}],"event":{"name":"CHI '23: CHI Conference on Human Factors in Computing Systems","location":"Hamburg Germany","acronym":"CHI '23","sponsor":["SIGCHI ACM Special Interest Group on Computer-Human Interaction"]},"container-title":["Proceedings of the 2023 CHI Conference on Human Factors in Computing Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3544548.3581006","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3544548.3581006","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T16:47:55Z","timestamp":1750178875000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3544548.3581006"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,4,19]]},"references-count":69,"alternative-id":["10.1145\/3544548.3581006","10.1145\/3544548"],"URL":"https:\/\/doi.org\/10.1145\/3544548.3581006","relation":{},"subject":[],"published":{"date-parts":[[2023,4,19]]},"assertion":[{"value":"2023-04-19","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}