{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,15]],"date-time":"2025-10-15T00:31:07Z","timestamp":1760488267854,"version":"build-2065373602"},"publisher-location":"New York, NY, USA","reference-count":22,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,11,28]]},"DOI":"10.1145\/3732437.3732768","type":"proceedings-article","created":{"date-parts":[[2025,10,14]],"date-time":"2025-10-14T10:33:54Z","timestamp":1760438034000},"page":"74-80","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Generating Spatially-Aware Dense Video Captions for Indoor Human Behavior Analysis with Position-Based Scene Knowledge"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-0226-3477","authenticated-orcid":false,"given":"Bin","family":"Chen","sequence":"first","affiliation":[{"name":"Department of Information Science and Technology, Kyushu university, Fukuoka, Fukuoka, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8834-5323","authenticated-orcid":false,"given":"Yugo","family":"Nakamura","sequence":"additional","affiliation":[{"name":"Department of Information Science and Technology, Kyushu university, Fukuoka, Fukuoka, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1362-8078","authenticated-orcid":false,"given":"Shogo","family":"Fukushima","sequence":"additional","affiliation":[{"name":"Department of Information Science and Technology, Kyushu university, Fukuoka, Fukuoka, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7156-9160","authenticated-orcid":false,"given":"Yutaka","family":"Arakawa","sequence":"additional","affiliation":[{"name":"Department of Information Science and Technology, Kyushu university, Fukuoka, Fukuoka, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,14]]},"reference":[{"key":"e_1_3_3_1_2_2","unstructured":"Badr AlKhamissi Millicent Li Asli Celikyilmaz Mona Diab and Marjan Ghazvininejad. 2022. A review on language models as knowledge bases. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2204.06031 (2022)."},{"key":"e_1_3_3_1_3_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW54120.2021.00356"},{"key":"e_1_3_3_1_4_2","first-page":"65","volume-title":"Proceedings of the acl workshop on intrinsic and extrinsic evaluation measures for machine translation and\/or summarization","author":"Banerjee Satanjeev","year":"2005","unstructured":"Satanjeev Banerjee and Alon Lavie. 2005. METEOR: An automatic metric for MT evaluation with improved correlation with human judgments. In Proceedings of the acl workshop on intrinsic and extrinsic evaluation measures for machine translation and\/or summarization. 65\u201372."},{"key":"e_1_3_3_1_5_2","doi-asserted-by":"crossref","first-page":"423","DOI":"10.1109\/ICRCA60878.2024.10649311","volume-title":"2024 8th International Conference on Robotics, Control and Automation (ICRCA)","author":"Bin CHEN","year":"2024","unstructured":"CHEN Bin, Yugo NAKAMURA, Shogo FUKUSHIMA, and Yutaka ARAKAWA. 2024. A Space Information-Enhanced Dense Video Caption for Indoor Human Action Recognition. In 2024 8th International Conference on Robotics, Control and Automation (ICRCA). IEEE, 423\u2013427."},{"key":"e_1_3_3_1_6_2","doi-asserted-by":"crossref","unstructured":"Bin Chen Yugo Nakamura Shogo Fukushima and Yutaka Arakawa. 2024. Retrieval-augmented-generation-enhanced Dense Video Caption for Human Indoor Activities: Disambiguating Caption Using Spatial Information Beyond Field of View Constraints. Sensors and Materials 36 10 (2024) 4549\u20134566.","DOI":"10.18494\/SAM5238"},{"key":"e_1_3_3_1_7_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.261"},{"key":"e_1_3_3_1_8_2","unstructured":"David Dale Elena Voita Lo\u00efc Barrault and Marta\u00a0R Costa-juss\u00e0. 2022. Detecting and mitigating hallucinations in machine translation: Model internal workings alone do well sentence similarity even better. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2212.08597 (2022)."},{"key":"e_1_3_3_1_9_2","first-page":"202","volume-title":"Medical Image Computing and Computer Assisted Intervention\u2013MICCAI 2019: 22nd International Conference, Shenzhen, China, October 13\u201317, 2019, Proceedings, Part III 22","author":"Ding Zhipeng","year":"2019","unstructured":"Zhipeng Ding, Xu Han, and Marc Niethammer. 2019. Votenet: A deep learning label fusion method for multi-atlas segmentation. In Medical Image Computing and Computer Assisted Intervention\u2013MICCAI 2019: 22nd International Conference, Shenzhen, China, October 13\u201317, 2019, Proceedings, Part III 22. Springer, 202\u2013210."},{"key":"e_1_3_3_1_10_2","first-page":"958","volume-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition workshops","author":"Iashin Vladimir","year":"2020","unstructured":"Vladimir Iashin and Esa Rahtu. 2020. Multi-modal dense video captioning. In Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition workshops. 958\u2013959."},{"key":"e_1_3_3_1_11_2","unstructured":"Patrick Lewis Ethan Perez Aleksandra Piktus Fabio Petroni Vladimir Karpukhin Naman Goyal Heinrich K\u00fcttler Mike Lewis Wen-tau Yih Tim Rockt\u00e4schel et\u00a0al. 2020. Retrieval-augmented generation for knowledge-intensive nlp tasks. Advances in Neural Information Processing Systems 33 (2020) 9459\u20139474."},{"key":"e_1_3_3_1_12_2","unstructured":"Zehan Li Xin Zhang Yanzhao Zhang Dingkun Long Pengjun Xie and Meishan Zhang. 2023. Towards general text embeddings with multi-stage contrastive learning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2308.03281 (2023)."},{"key":"e_1_3_3_1_13_2","unstructured":"Yinhan Liu. 2019. Roberta: A robustly optimized bert pretraining approach. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1907.11692 364 (2019)."},{"key":"e_1_3_3_1_14_2","unstructured":"Long Ouyang Jeffrey Wu Xu Jiang Diogo Almeida Carroll Wainwright Pamela Mishkin Chong Zhang Sandhini Agarwal Katarina Slama Alex Ray et\u00a0al. 2022. Training language models to follow instructions with human feedback. Advances in neural information processing systems 35 (2022) 27730\u201327744."},{"key":"e_1_3_3_1_15_2","unstructured":"Gemini Team Rohan Anil Sebastian Borgeaud Yonghui Wu Jean-Baptiste Alayrac Jiahui Yu Radu Soricut Johan Schalkwyk Andrew\u00a0M Dai Anja Hauth et\u00a0al. 2023. Gemini: a family of highly capable multimodal models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2312.11805 (2023)."},{"key":"e_1_3_3_1_16_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.510"},{"key":"e_1_3_3_1_17_2","doi-asserted-by":"crossref","unstructured":"Yunbin Tu Chang Zhou Junjun Guo Shengxiang Gao and Zhengtao Yu. 2021. Enhancing the alignment between target words and corresponding frames for video captioning. Pattern Recognition 111 (2021) 107702.","DOI":"10.1016\/j.patcog.2020.107702"},{"key":"e_1_3_3_1_18_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00751"},{"key":"e_1_3_3_1_19_2","unstructured":"Teng Wang Jinrui Zhang Feng Zheng Wenhao Jiang Ran Cheng and Ping Luo. 2023. Learning grounded vision-language representation for versatile understanding in untrimmed videos. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2303.06378 (2023)."},{"key":"e_1_3_3_1_20_2","doi-asserted-by":"crossref","unstructured":"Teng Wang Huicheng Zheng Mingjing Yu Qian Tian and Haifeng Hu. 2020. Event-centric hierarchical representation for dense video captioning. IEEE Transactions on Circuits and Systems for Video Technology 31 5 (2020) 1890\u20131900.","DOI":"10.1109\/TCSVT.2020.3014606"},{"key":"e_1_3_3_1_21_2","unstructured":"Shangyu Wu Ying Xiong Yufei Cui Haolun Wu Can Chen Ye Yuan Lianming Huang Xue Liu Tei-Wei Kuo Nan Guan et\u00a0al. 2024. Retrieval-Augmented Generation for Natural Language Processing: A Survey. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2407.13193 (2024)."},{"key":"e_1_3_3_1_22_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01032"},{"key":"e_1_3_3_1_23_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01589"}],"event":{"name":"ICEA 2024: The 2024 International Conference on Intelligent Computing and its Emerging Applicaton","location":"Tokyo Japan","acronym":"ICEA 2024"},"container-title":["Proceedings of the 2024 International Conference on Intelligent Computing and its Emerging Applicaton"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3732437.3732768","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,14]],"date-time":"2025-10-14T10:35:09Z","timestamp":1760438109000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3732437.3732768"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,28]]},"references-count":22,"alternative-id":["10.1145\/3732437.3732768","10.1145\/3732437"],"URL":"https:\/\/doi.org\/10.1145\/3732437.3732768","relation":{},"subject":[],"published":{"date-parts":[[2024,11,28]]},"assertion":[{"value":"2025-10-14","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}