{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,13]],"date-time":"2026-04-13T15:23:22Z","timestamp":1776093802291,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":126,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,29]],"date-time":"2024-10-29T00:00:00Z","timestamp":1730160000000},"content-version":"vor","delay-in-days":366,"URL":"http:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/100000001","name":"NSF (National Science Foundation)","doi-asserted-by":"publisher","award":["CCF-2211428"],"award-info":[{"award-number":["CCF-2211428"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,10,29]]},"DOI":"10.1145\/3586183.3606776","type":"proceedings-article","created":{"date-parts":[[2023,10,20]],"date-time":"2023-10-20T20:46:22Z","timestamp":1697834782000},"page":"1-18","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":8,"title":["PEANUT: A Human-AI Collaborative Tool for Annotating Audio-Visual Data"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-7040-2326","authenticated-orcid":false,"given":"Zheng","family":"Zhang","sequence":"first","affiliation":[{"name":"Department of Computer Science and Engineering, University of Notre Dame, United States and Department of Computer Science and Engineering, University of Notre Dame, United States"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7374-7453","authenticated-orcid":false,"given":"Zheng","family":"Ning","sequence":"additional","affiliation":[{"name":"Department of Computer Science and Engineering, University of Notre Dame, United States and Department of Computer Science and Engineering, University of Notre Dame, United States"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2183-822X","authenticated-orcid":false,"given":"Chenliang","family":"Xu","sequence":"additional","affiliation":[{"name":"Department of Computer Science, University of Rochester, United States and Department of Computer Science, University of Rochester, United States"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1423-4513","authenticated-orcid":false,"given":"Yapeng","family":"Tian","sequence":"additional","affiliation":[{"name":"Department of Computer Science, University of Texas at Dallas, United States and Department of Computer Science, University of Texas at Dallas, United States"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7902-7625","authenticated-orcid":false,"given":"Toby Jia-Jun","family":"Li","sequence":"additional","affiliation":[{"name":"Department of Computer Science and Engineering, University of Notre Dame, United States and Department of Computer Science and Engineering, University of Notre Dame, United States"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,10,29]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"Attenuation of Human Bias in Artificial Intelligence: An Exploratory Approach. 2021 6th International Conference on Inventive Computation Technologies (ICICT)","author":"Ahmed Saad\u00a0Bin","year":"2021","unstructured":"Saad\u00a0Bin Ahmed, Saif\u00a0Ali Athyaab, and Shaik\u00a0Abdul Muqtadeer. 2021. Attenuation of Human Bias in Artificial Intelligence: An Exploratory Approach. 2021 6th International Conference on Inventive Computation Technologies (ICICT) (2021), 557\u2013563."},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00774"},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00033"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1145\/3290605.3300233"},{"key":"e_1_3_2_2_5_1","volume-title":"Proceedings of the British Machine Vision Conference (BMVC).","author":"Anayurt Hazan","year":"2019","unstructured":"Hazan Anayurt, Sezai\u00a0Artun Ozyegin, Ulfet Cetin, Utku Aktas, and Sinan Kalkan. 2019. Searching for Ambiguous Objects in Videos using Relational Referring Expressions. In Proceedings of the British Machine Vision Conference (BMVC)."},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.279"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.73"},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"crossref","unstructured":"Relja Arandjelovic and Andrew Zisserman. 2018. Objects that sound. In ECCV.","DOI":"10.1007\/978-3-030-01246-5_27"},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3449163"},{"key":"e_1_3_2_2_10_1","volume-title":"Soundnet: Learning sound representations from unlabeled video. Advances in neural information processing systems 29","author":"Aytar Yusuf","year":"2016","unstructured":"Yusuf Aytar, Carl Vondrick, and Antonio Torralba. 2016. Soundnet: Learning sound representations from unlabeled video. Advances in neural information processing systems 29 (2016), 892\u2013900."},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/P14-1133"},{"key":"e_1_3_2_2_12_1","volume-title":"Ilastik: interactive machine learning for (bio) image analysis. Nature methods 16, 12","author":"Berg Stuart","year":"2019","unstructured":"Stuart Berg, Dominik Kutra, Thorben Kroeger, Christoph\u00a0N Straehle, Bernhard\u00a0X Kausler, Carsten Haubold, Martin Schiegg, Janez Ales, Thorsten Beier, Markus Rudy, 2019. Ilastik: interactive machine learning for (bio) image analysis. Nature methods 16, 12 (2019), 1226\u20131232."},{"key":"e_1_3_2_2_13_1","volume-title":"Exposing and Correcting the Gender Bias in Image Captioning Datasets and Models. ArXiv abs\/1912.00578","author":"Bhargava Shruti","year":"2019","unstructured":"Shruti Bhargava and David Forsyth. 2019. Exposing and Correcting the Gender Bias in Image Captioning Datasets and Models. ArXiv abs\/1912.00578 (2019)."},{"key":"e_1_3_2_2_14_1","volume-title":"Algorithmic Factors Influencing Bias in Machine Learning. In PKDD\/ECML Workshops.","author":"Blanzeisky William","year":"2021","unstructured":"William Blanzeisky and Padraig Cunningham. 2021. Algorithmic Factors Influencing Bias in Machine Learning. In PKDD\/ECML Workshops."},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.3390\/engproc2021007039"},{"key":"e_1_3_2_2_16_1","volume-title":"NIPS workshop on computational social science and the wisdom of crowds.","author":"Brew Anthony","year":"2010","unstructured":"Anthony Brew, Derek Greene, and P\u00e1draig Cunningham. 2010. The interaction between supervised learning and crowdsourcing. In NIPS workshop on computational social science and the wisdom of crowds."},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3290605.3300234"},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-94-024-0881-2_7"},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1145\/3411764.3445165"},{"key":"e_1_3_2_2_20_1","volume-title":"Learning to set waypoints for audio-visual navigation. arXiv preprint arXiv:2008.09622","author":"Chen Changan","year":"2020","unstructured":"Changan Chen, Sagnik Majumder, Ziad Al-Halah, Ruohan Gao, Santhosh\u00a0Kumar Ramakrishnan, and Kristen Grauman. 2020. Learning to set waypoints for audio-visual navigation. arXiv preprint arXiv:2008.09622 (2020)."},{"key":"e_1_3_2_2_21_1","volume-title":"Understanding and Mitigating Annotation Bias in Facial Expression Recognition. 2021 IEEE\/CVF International Conference on Computer Vision (ICCV)","author":"Chen Yunliang","year":"2021","unstructured":"Yunliang Chen and Jungseock Joo. 2021. Understanding and Mitigating Annotation Bias in Facial Expression Recognition. 2021 IEEE\/CVF International Conference on Computer Vision (ICCV) (2021), 14960\u201314971."},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/P15-1017"},{"key":"e_1_3_2_2_23_1","volume-title":"Medical Imaging 2022: Computer-Aided Diagnosis, Vol.\u00a012033","author":"Choi Youngwon","unstructured":"Youngwon Choi, Marlena Garcia, Steven\u00a0S Raman, Dieter\u00a0R Enzmann, and Matthew\u00a0S Brown. 2022. AI-human interactive pipeline with feedback to accelerate medical image annotation. In Medical Imaging 2022: Computer-Aided Diagnosis, Vol.\u00a012033. SPIE, 741\u2013747."},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"publisher","DOI":"10.5555\/1622737.1622744"},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1038\/nature09304"},{"key":"e_1_3_2_2_26_1","volume-title":"A Farewell to the Bias-Variance Tradeoff? An Overview of the Theory of Overparameterized Machine Learning. ArXiv abs\/2109.02355","author":"Dar Yehuda","year":"2021","unstructured":"Yehuda Dar, Vidya Muthukumar, and Richard Baraniuk. 2021. A Farewell to the Bias-Variance Tradeoff? An Overview of the Theory of Overparameterized Machine Learning. ArXiv abs\/2109.02355 (2021)."},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11042-012-1352-1"},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/3397481.3450698"},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.167"},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"publisher","DOI":"10.1145\/3343031.3350535"},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3197517.3201357"},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.findings-emnlp.181"},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDMW.2019.00161"},{"key":"e_1_3_2_2_34_1","volume-title":"Toby Jia-Jun Li, and Simon\u00a0Tangi Perrault","author":"Gao Jie","year":"2023","unstructured":"Jie Gao, Yuchen Guo, Gionnieve Lim, Tianqin Zhan, Zheng Zhang, Toby Jia-Jun Li, and Simon\u00a0Tangi Perrault. 2023. CollabCoder: A GPT-Powered Workflow for Collaborative Qualitative Analysis. arXiv preprint arXiv:2304.07366 (2023)."},{"key":"e_1_3_2_2_35_1","volume-title":"ObjectFolder: A Dataset of Objects with Implicit Visual, Auditory, and Tactile Representations. arXiv preprint arXiv:2109.07991","author":"Gao Ruohan","year":"2021","unstructured":"Ruohan Gao, Yen-Yu Chang, Shivani Mall, Li Fei-Fei, and Jiajun Wu. 2021. ObjectFolder: A Dataset of Objects with Implicit Visual, Auditory, and Tactile Representations. arXiv preprint arXiv:2109.07991 (2021)."},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01219-9_3"},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00041"},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00398"},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/3544548.3581352"},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1145\/2047196.2047205"},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/MIS.2009.36"},{"key":"e_1_3_2_2_42_1","volume-title":"International Journal of Image and Graphics","author":"Hamroun Mohamed","year":"2021","unstructured":"Mohamed Hamroun, Karim Tamine, and Beno\u00eet Crespin. 2021. Multimodal Video Indexing (MVI): A New Method Based on Machine Learning and Semi-Automatic Annotation on Large Video Collections. International Journal of Image and Graphics (2021), 2250022."},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2020.106622"},{"key":"e_1_3_2_2_44_1","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.1807184115"},{"key":"e_1_3_2_2_45_1","unstructured":"John\u00a0R Hershey and Javier\u00a0R Movellan. 2000. Audio vision: Using audio-visual synchrony to locate sounds. In Advances in neural information processing systems. 813\u2013819."},{"key":"e_1_3_2_2_46_1","volume-title":"Deep neural networks for acoustic modeling in speech recognition: The shared views of four research groups","author":"Hinton Geoffrey","year":"2012","unstructured":"Geoffrey Hinton, Li Deng, Dong Yu, George\u00a0E Dahl, Abdel-rahman Mohamed, Navdeep Jaitly, Andrew Senior, Vincent Vanhoucke, Patrick Nguyen, Tara\u00a0N Sainath, 2012. Deep neural networks for acoustic modeling in speech recognition: The shared views of four research groups. IEEE Signal processing magazine 29, 6 (2012), 82\u201397."},{"key":"e_1_3_2_2_47_1","doi-asserted-by":"publisher","DOI":"10.1145\/302979.303030"},{"key":"e_1_3_2_2_48_1","doi-asserted-by":"crossref","unstructured":"Di Hu Feiping Nie and Xuelong Li. 2019. Deep Multimodal Clustering for Unsupervised Audiovisual Learning. In CVPR.","DOI":"10.1109\/CVPR.2019.00947"},{"key":"e_1_3_2_2_49_1","volume-title":"Advances in Neural Information Processing Systems, H.\u00a0Larochelle, M.\u00a0Ranzato, R.\u00a0Hadsell, M.F. Balcan, and H.\u00a0Lin (Eds.). Vol.\u00a033. Curran Associates","author":"Hu Di","year":"2020","unstructured":"Di Hu, Rui Qian, Minyue Jiang, Xiao Tan, Shilei Wen, Errui Ding, Weiyao Lin, and Dejing Dou. 2020. Discriminative Sounding Objects Localization via Self-supervised Audiovisual Matching. In Advances in Neural Information Processing Systems, H.\u00a0Larochelle, M.\u00a0Ranzato, R.\u00a0Hadsell, M.F. Balcan, and H.\u00a0Lin (Eds.). Vol.\u00a033. Curran Associates, Inc., 10077\u201310087. https:\/\/proceedings.neurips.cc\/paper\/2020\/file\/7288251b27c8f0e73f4d7f483b06a785-Paper.pdf"},{"key":"e_1_3_2_2_50_1","volume-title":"Class-aware sounding objects localization via audiovisual correspondence","author":"Hu Di","year":"2021","unstructured":"Di Hu, Yake Wei, Rui Qian, Weiyao Lin, Ruihua Song, and Ji-Rong Wen. 2021. Class-aware sounding objects localization via audiovisual correspondence. IEEE Transactions on Pattern Analysis and Machine Intelligence (2021)."},{"key":"e_1_3_2_2_51_1","volume-title":"Artificial intelligence and the future of work: Human-AI symbiosis in organizational decision making. Business horizons 61, 4","author":"Jarrahi Mohammad\u00a0Hossein","year":"2018","unstructured":"Mohammad\u00a0Hossein Jarrahi. 2018. Artificial intelligence and the future of work: Human-AI symbiosis in organizational decision making. Business horizons 61, 4 (2018), 577\u2013586."},{"key":"e_1_3_2_2_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/TVCG.2012.219"},{"key":"e_1_3_2_2_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01384"},{"key":"e_1_3_2_2_54_1","volume-title":"Artificial intelligence-aided clinical annotation of a large multi-cancer genomic dataset. Nature communications 12, 1","author":"Kehl L","year":"2021","unstructured":"Kenneth\u00a0L Kehl, Wenxin Xu, Alexander Gusev, Ziad Bakouny, Toni\u00a0K Choueiri, Irbaz\u00a0Bin Riaz, Haitham Elmarakeby, Eliezer\u00a0M Van\u00a0Allen, and Deborah Schrag. 2021. Artificial intelligence-aided clinical annotation of a large multi-cancer genomic dataset. Nature communications 12, 1 (2021), 7304."},{"key":"e_1_3_2_2_55_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2005.274"},{"key":"e_1_3_2_2_56_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-2041"},{"key":"e_1_3_2_2_57_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2020.3030497"},{"key":"e_1_3_2_2_58_1","volume-title":"Advances in experimental social psychology. Vol.\u00a022","author":"Langer J","unstructured":"Ellen\u00a0J Langer. 1989. Minding matters: The consequences of mindlessness\u2013mindfulness. In Advances in experimental social psychology. Vol.\u00a022. Elsevier, 137\u2013173."},{"key":"e_1_3_2_2_59_1","doi-asserted-by":"publisher","DOI":"10.1145\/2702123.2702416"},{"key":"e_1_3_2_2_60_1","volume-title":"Mitigating Gender Bias in Machine Learning Data Sets. ArXiv abs\/2005.06898","author":"Leavy Susan","year":"2020","unstructured":"Susan Leavy, Gerardine Meaney, Karen Wade, and Derek Greene. 2020. Mitigating Gender Bias in Machine Learning Data Sets. ArXiv abs\/2005.06898 (2020)."},{"key":"e_1_3_2_2_61_1","doi-asserted-by":"publisher","DOI":"10.1145\/3025453.3025483"},{"key":"e_1_3_2_2_62_1","doi-asserted-by":"publisher","DOI":"10.1145\/3379337.3415820"},{"key":"e_1_3_2_2_63_1","doi-asserted-by":"publisher","DOI":"10.1145\/3332165.3347899"},{"key":"e_1_3_2_2_64_1","volume-title":"Man-computer symbiosis. IRE transactions on human factors in electronics1","author":"Licklider CR","year":"1960","unstructured":"Joseph\u00a0CR Licklider. 1960. Man-computer symbiosis. IRE transactions on human factors in electronics1 (1960), 4\u201311."},{"key":"e_1_3_2_2_65_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.106"},{"key":"e_1_3_2_2_66_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"e_1_3_2_2_67_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683226"},{"key":"e_1_3_2_2_68_1","unstructured":"Minzhe Liu Li Du Yuan Du Ruofan Guo and Xiaoliang Chen. 2020. Faster Human-Machine Collaboration Bounding Box Annotation Framework Based on Active Learning. (2020)."},{"key":"e_1_3_2_2_69_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i04.5926"},{"key":"e_1_3_2_2_70_1","volume-title":"Use what you have: Video retrieval using representations from collaborative experts. arXiv preprint arXiv:1907.13487","author":"Liu Yang","year":"2019","unstructured":"Yang Liu, Samuel Albanie, Arsha Nagrani, and Andrew Zisserman. 2019. Use what you have: Video retrieval using representations from collaborative experts. arXiv preprint arXiv:1907.13487 (2019)."},{"key":"e_1_3_2_2_71_1","doi-asserted-by":"publisher","DOI":"10.1145\/3313831.3376739"},{"key":"e_1_3_2_2_72_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2018.06.003"},{"key":"e_1_3_2_2_73_1","doi-asserted-by":"publisher","DOI":"10.1145\/2702123.2702553"},{"key":"e_1_3_2_2_74_1","volume-title":"ETHOS: a multi-label hate speech detection dataset. Complex & Intelligent Systems","author":"Mollas Ioannis","year":"2020","unstructured":"Ioannis Mollas, Zoe Chrysopoulou, Stamatis Karlos, and Grigorios Tsoumakas. 2020. ETHOS: a multi-label hate speech detection dataset. Complex & Intelligent Systems (2020), 1\u201316."},{"key":"e_1_3_2_2_75_1","volume-title":"Self-supervised generation of spatial audio for 360 video. arXiv preprint arXiv:1809.02587","author":"Morgado Pedro","year":"2018","unstructured":"Pedro Morgado, Nuno Vasconcelos, Timothy Langlois, and Oliver Wang. 2018. Self-supervised generation of spatial audio for 360 video. arXiv preprint arXiv:1809.02587 (2018)."},{"key":"e_1_3_2_2_76_1","doi-asserted-by":"publisher","DOI":"10.1145\/3290605.3300356"},{"key":"e_1_3_2_2_77_1","doi-asserted-by":"publisher","DOI":"10.1145\/3411764.3445402"},{"key":"e_1_3_2_2_78_1","doi-asserted-by":"crossref","unstructured":"Micah\u00a0M Murray and Mark\u00a0T Wallace. 2011. The neural bases of multisensory processes. (2011).","DOI":"10.1201\/9781439812174"},{"key":"e_1_3_2_2_79_1","volume-title":"A survey on annotation tools for the biomedical literature. Briefings in bioinformatics 15, 2","author":"Neves Mariana","year":"2014","unstructured":"Mariana Neves and Ulf Leser. 2014. A survey on annotation tools for the biomedical literature. Briefings in bioinformatics 15, 2 (2014), 327\u2013340."},{"key":"e_1_3_2_2_80_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581641.3584067"},{"key":"e_1_3_2_2_81_1","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2017.2713783"},{"key":"e_1_3_2_2_82_1","doi-asserted-by":"publisher","DOI":"10.1145\/302979.303163"},{"key":"e_1_3_2_2_83_1","doi-asserted-by":"publisher","DOI":"10.1145\/319382.319398"},{"key":"e_1_3_2_2_84_1","doi-asserted-by":"publisher","DOI":"10.1145\/330534.330538"},{"key":"e_1_3_2_2_85_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01231-1_39"},{"key":"e_1_3_2_2_86_1","doi-asserted-by":"publisher","DOI":"10.1145\/3379337.3415864"},{"key":"e_1_3_2_2_87_1","doi-asserted-by":"crossref","unstructured":"Rui Qian Di Hu Heinrich Dinkel Mengyue Wu Ning Xu and Weiyao Lin. 2020. Multiple Sound Sources Localization from Coarse to Fine. In ECCV.","DOI":"10.1007\/978-3-030-58565-5_18"},{"key":"e_1_3_2_2_88_1","doi-asserted-by":"crossref","unstructured":"Nan Qiao Yuyin Sun Chongyu Liu Lu Xia Jiajia Luo K. Zhang and Cheng-Hao Kuo. 2022. Human-in-the-Loop Video Semantic Segmentation Auto-Annotation.","DOI":"10.1109\/WACV56688.2023.00583"},{"key":"e_1_3_2_2_89_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00900"},{"key":"e_1_3_2_2_90_1","doi-asserted-by":"publisher","DOI":"10.1145\/502716.502737"},{"key":"e_1_3_2_2_91_1","doi-asserted-by":"publisher","DOI":"10.14778\/3157794.3157797"},{"key":"e_1_3_2_2_92_1","volume-title":"Faster r-cnn: Towards real-time object detection with region proposal networks. Advances in neural information processing systems 28","author":"Ren Shaoqing","year":"2015","unstructured":"Shaoqing Ren, Kaiming He, Ross Girshick, and Jian Sun. 2015. Faster r-cnn: Towards real-time object detection with region proposal networks. Advances in neural information processing systems 28 (2015), 91\u201399."},{"key":"e_1_3_2_2_93_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV48630.2021.00009"},{"key":"e_1_3_2_2_94_1","doi-asserted-by":"publisher","DOI":"10.1145\/3411764.3445591"},{"key":"e_1_3_2_2_95_1","doi-asserted-by":"publisher","DOI":"10.1145\/3411764.3445518"},{"key":"e_1_3_2_2_96_1","doi-asserted-by":"publisher","DOI":"10.1145\/564376.564421"},{"key":"e_1_3_2_2_97_1","volume-title":"Proceedings of the 2012 joint conference on empirical methods in natural language processing and computational natural language learning. 523\u2013534","author":"Schmitz Michael","year":"2012","unstructured":"Michael Schmitz, Stephen Soderland, Robert Bart, Oren Etzioni, 2012. Open language learning for information extraction. In Proceedings of the 2012 joint conference on empirical methods in natural language processing and computational natural language learning. 523\u2013534."},{"key":"e_1_3_2_2_98_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00458"},{"key":"e_1_3_2_2_99_1","doi-asserted-by":"publisher","DOI":"10.2200\/S00429ED1V01Y201207AIM018"},{"key":"e_1_3_2_2_100_1","volume-title":"Label Sleuth: From Unlabeled Text to a Classifier in a Few Hours. arXiv preprint arXiv:2208.01483","author":"Shnarch Eyal","year":"2022","unstructured":"Eyal Shnarch, Alon Halfon, Ariel Gera, Marina Danilevsky, Yannis Katsis, Leshem Choshen, Martin\u00a0Santillan Cooper, Dina Epelboim, Zheng Zhang, Dakuo Wang, 2022. Label Sleuth: From Unlabeled Text to a Classifier in a Few Hours. arXiv preprint arXiv:2208.01483 (2022)."},{"key":"e_1_3_2_2_101_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.emnlp-demos.16"},{"key":"e_1_3_2_2_102_1","volume-title":"A graph-to-sequence model for AMR-to-text generation. arXiv preprint arXiv:1805.02473","author":"Song Linfeng","year":"2018","unstructured":"Linfeng Song, Yue Zhang, Zhiguo Wang, and Daniel Gildea. 2018. A graph-to-sequence model for AMR-to-text generation. arXiv preprint arXiv:1805.02473 (2018)."},{"key":"e_1_3_2_2_103_1","volume-title":"An attempt towards interpretable audio-visual video captioning. arXiv preprint arXiv:1812.02872","author":"Tian Yapeng","year":"2018","unstructured":"Yapeng Tian, Chenxiao Guan, Justin Goodman, Marc Moore, and Chenliang Xu. 2018. An attempt towards interpretable audio-visual video captioning. arXiv preprint arXiv:1812.02872 (2018)."},{"key":"e_1_3_2_2_104_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00277"},{"key":"e_1_3_2_2_105_1","volume-title":"Proceedings, Part III 16","author":"Tian Yapeng","year":"2020","unstructured":"Yapeng Tian, Dingzeyu Li, and Chenliang Xu. 2020. Unified multisensory perception: Weakly-supervised audio-visual video parsing. In Computer Vision\u2013ECCV 2020: 16th European Conference, Glasgow, UK, August 23\u201328, 2020, Proceedings, Part III 16. Springer, 436\u2013454."},{"key":"e_1_3_2_2_106_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01216-8_16"},{"key":"e_1_3_2_2_107_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patrec.2015.02.001"},{"key":"e_1_3_2_2_108_1","volume-title":"recaptcha: Human-based character recognition via web security measures. Science 321, 5895","author":"Von\u00a0Ahn Luis","year":"2008","unstructured":"Luis Von\u00a0Ahn, Benjamin Maurer, Colin McMillen, David Abraham, and Manuel Blum. 2008. recaptcha: Human-based character recognition via web security measures. Science 321, 5895 (2008), 1465\u20131468."},{"key":"e_1_3_2_2_109_1","unstructured":"Kentaro Wada. 2016. labelme: Image Polygonal Annotation with Python. https:\/\/github.com\/wkentaro\/labelme."},{"key":"e_1_3_2_2_110_1","doi-asserted-by":"publisher","DOI":"10.1145\/3411764.3445526"},{"key":"e_1_3_2_2_111_1","doi-asserted-by":"publisher","DOI":"10.1145\/3411763.3450394"},{"key":"e_1_3_2_2_112_1","doi-asserted-by":"publisher","DOI":"10.1145\/3359313"},{"key":"e_1_3_2_2_113_1","doi-asserted-by":"publisher","DOI":"10.1145\/3411764.3445347"},{"key":"e_1_3_2_2_114_1","doi-asserted-by":"publisher","DOI":"10.1145\/3025453.3025768"},{"key":"e_1_3_2_2_115_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.acl-long.523"},{"key":"e_1_3_2_2_116_1","unstructured":"Yuxin Wu Alexander Kirillov Francisco Massa Wan-Yen Lo and Ross Girshick. 2019. Detectron2. https:\/\/github.com\/facebookresearch\/detectron2."},{"key":"e_1_3_2_2_117_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00138"},{"key":"e_1_3_2_2_118_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00639"},{"key":"e_1_3_2_2_119_1","unstructured":"Dean Wyatte. 2019. De-biasing Weakly Supervised Learning by Regularizing Prediction Entropy. (2019)."},{"key":"e_1_3_2_2_120_1","volume-title":"Addressing Training Bias via Automated Image Annotation. arXiv: Computer Vision and Pattern Recognition","author":"Xiao Zhujun","year":"2018","unstructured":"Zhujun Xiao, Yanzi Zhu, Yuxin Chen, Ben\u00a0Y. Zhao, Junchen Jiang, and Haitao Zheng. 2018. Addressing Training Bias via Automated Image Annotation. arXiv: Computer Vision and Pattern Recognition (2018)."},{"key":"e_1_3_2_2_121_1","unstructured":"Xtract.io. 2020. Xtract.io video annotation tool. https:\/\/www.xtract.io\/lp\/image-annotation-tool"},{"key":"e_1_3_2_2_122_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISM.2018.00-21"},{"key":"e_1_3_2_2_123_1","volume-title":"OneLabeler: A Flexible System for Building Data Labeling Tools. In CHI Conference on Human Factors in Computing Systems. 1\u201322","author":"Zhang Yu","year":"2022","unstructured":"Yu Zhang, Yun Wang, Haidong Zhang, Bin Zhu, Siming Chen, and Dongmei Zhang. 2022. OneLabeler: A Flexible System for Building Data Labeling Tools. In CHI Conference on Human Factors in Computing Systems. 1\u201322."},{"key":"e_1_3_2_2_124_1","volume-title":"VISAR: A Human-AI Argumentative Writing Assistant with Visual Programming and Rapid Draft Prototyping. arXiv preprint arXiv:2304.07810","author":"Zhang Zheng","year":"2023","unstructured":"Zheng Zhang, Jie Gao, Ranjodh\u00a0Singh Dhaliwal, and Toby Jia-Jun Li. 2023. VISAR: A Human-AI Argumentative Writing Assistant with Visual Programming and Rapid Draft Prototyping. arXiv preprint arXiv:2304.07810 (2023)."},{"key":"e_1_3_2_2_125_1","doi-asserted-by":"crossref","unstructured":"Hang Zhao Chuang Gan Andrew Rouditchenko Carl Vondrick Josh McDermott and Antonio Torralba. 2018. The sound of pixels. In ECCV.","DOI":"10.1007\/978-3-030-01246-5_35"},{"key":"e_1_3_2_2_126_1","volume-title":"A brief introduction to weakly supervised learning. National science review 5, 1","author":"Zhou Zhi-Hua","year":"2018","unstructured":"Zhi-Hua Zhou. 2018. A brief introduction to weakly supervised learning. National science review 5, 1 (2018), 44\u201353."}],"event":{"name":"UIST '23: The 36th Annual ACM Symposium on User Interface Software and Technology","location":"San Francisco CA USA","acronym":"UIST '23","sponsor":["SIGGRAPH ACM Special Interest Group on Computer Graphics and Interactive Techniques","SIGCHI ACM Special Interest Group on Computer-Human Interaction"]},"container-title":["Proceedings of the 36th Annual ACM Symposium on User Interface Software and Technology"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3586183.3606776","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3586183.3606776","content-type":"application\/pdf","content-version":"vor","intended-application":"syndication"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3586183.3606776","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T23:53:18Z","timestamp":1755820398000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3586183.3606776"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,29]]},"references-count":126,"alternative-id":["10.1145\/3586183.3606776","10.1145\/3586183"],"URL":"https:\/\/doi.org\/10.1145\/3586183.3606776","relation":{},"subject":[],"published":{"date-parts":[[2023,10,29]]},"assertion":[{"value":"2023-10-29","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}