{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,2]],"date-time":"2026-06-02T09:18:13Z","timestamp":1780391893545,"version":"3.54.1"},"reference-count":35,"publisher":"Wiley","license":[{"start":{"date-parts":[[2026,6,2]],"date-time":"2026-06-02T00:00:00Z","timestamp":1780358400000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by\/4.0\/"},{"start":{"date-parts":[[2026,6,2]],"date-time":"2026-06-02T00:00:00Z","timestamp":1780358400000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/doi.wiley.com\/10.1002\/tdm_license_1.1"}],"funder":[{"DOI":"10.13039\/501100003593","name":"Conselho Nacional de Desenvolvimento Cient\u00edfico e Tecnol\u00f3gico","doi-asserted-by":"publisher","award":["CNPq"],"award-info":[{"award-number":["CNPq"]}],"id":[{"id":"10.13039\/501100003593","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003593","name":"Conselho Nacional de Desenvolvimento Cient\u00edfico e Tecnol\u00f3gico","doi-asserted-by":"publisher","award":["#311144\/2022\u20105"],"award-info":[{"award-number":["#311144\/2022\u20105"]}],"id":[{"id":"10.13039\/501100003593","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003593","name":"Conselho Nacional de Desenvolvimento Cient\u00edfico e Tecnol\u00f3gico","doi-asserted-by":"publisher","award":["#132348\/2025\u20100"],"award-info":[{"award-number":["#132348\/2025\u20100"]}],"id":[{"id":"10.13039\/501100003593","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003593","name":"Conselho Nacional de Desenvolvimento Cient\u00edfico e Tecnol\u00f3gico","doi-asserted-by":"publisher","award":["#132349\/2025\u20106"],"award-info":[{"award-number":["#132349\/2025\u20106"]}],"id":[{"id":"10.13039\/501100003593","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100004586","name":"Funda\u00e7\u00e3o Carlos Chagas Filho de Amparo \u00e0 Pesquisa do Estado do Rio de Janeiro","doi-asserted-by":"publisher","award":["FAPERJ"],"award-info":[{"award-number":["FAPERJ"]}],"id":[{"id":"10.13039\/501100004586","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100004586","name":"Funda\u00e7\u00e3o Carlos Chagas Filho de Amparo \u00e0 Pesquisa do Estado do Rio de Janeiro","doi-asserted-by":"publisher","award":["#E\u201026\/210.585\/2025"],"award-info":[{"award-number":["#E\u201026\/210.585\/2025"]}],"id":[{"id":"10.13039\/501100004586","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001807","name":"Funda\u00e7\u00e3o de Amparo \u00e0 Pesquisa do Estado de S\u00e3o Paulo","doi-asserted-by":"publisher","award":["FAPESP"],"award-info":[{"award-number":["FAPESP"]}],"id":[{"id":"10.13039\/501100001807","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001807","name":"Funda\u00e7\u00e3o de Amparo \u00e0 Pesquisa do Estado de S\u00e3o Paulo","doi-asserted-by":"publisher","award":["#2021\/07012\u20100"],"award-info":[{"award-number":["#2021\/07012\u20100"]}],"id":[{"id":"10.13039\/501100001807","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001807","name":"Funda\u00e7\u00e3o de Amparo \u00e0 Pesquisa do Estado de S\u00e3o Paulo","doi-asserted-by":"publisher","award":["#2023\/04868\u20107"],"award-info":[{"award-number":["#2023\/04868\u20107"]}],"id":[{"id":"10.13039\/501100001807","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["NSF"],"award-info":[{"award-number":["NSF"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000001","name":"National Science Foundation","doi-asserted-by":"publisher","award":["#OAC\u20102411221"],"award-info":[{"award-number":["#OAC\u20102411221"]}],"id":[{"id":"10.13039\/100000001","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000140","name":"U.S. Department of Transportation","doi-asserted-by":"publisher","award":["USDOT"],"award-info":[{"award-number":["USDOT"]}],"id":[{"id":"10.13039\/100000140","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/100000140","name":"U.S. Department of Transportation","doi-asserted-by":"publisher","award":["#69A3551747124"],"award-info":[{"award-number":["#69A3551747124"]}],"id":[{"id":"10.13039\/100000140","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100002322","name":"Coordena\u00e7\u00e3o de Aperfei\u00e7oamento de Pessoal de N\u00edvel Superior","doi-asserted-by":"publisher","award":["ROR identifier: 00x0ma614"],"award-info":[{"award-number":["ROR identifier: 00x0ma614"]}],"id":[{"id":"10.13039\/501100002322","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["onlinelibrary.wiley.com"],"crossmark-restriction":true},"short-container-title":["Computer Graphics Forum"],"abstract":"<jats:title>Abstract<\/jats:title>\n                  <jats:p>\n                    Extracting actionable insights from long\u2010duration urban videos is often labor\u2010intensive: analysts must manually sift through raw footage to pinpoint target events or uncover broader behavioral trends. In this work, we present\n                    <jats:italic>\n                      U\n                      <jats:sc>rban<\/jats:sc>\n                      C\n                      <jats:sc>lip<\/jats:sc>\n                      A\n                      <jats:sc>tlas<\/jats:sc>\n                    <\/jats:italic>\n                    , a visual analytics system for exploring long urban videos recorded at street intersections.\n                    <jats:italic>\n                      U\n                      <jats:sc>rban<\/jats:sc>\n                      C\n                      <jats:sc>lip<\/jats:sc>\n                      A\n                      <jats:sc>tlas<\/jats:sc>\n                    <\/jats:italic>\n                    combines retrieval\u2010augmented generation (RAG), taxonomy\u2010aware entity extraction, and video grounding to support event retrieval and interpretation. The system segments extended recordings into short clips, generates textual descriptions with a vision\u2013language model, and indexes them for semantic retrieval. A knowledge graph maps entities and relations from LLM answers onto a domain\u2010specific taxonomy and aligns them with detected objects and trajectories to support visual grounding and verification.\n                    <jats:italic>\n                      U\n                      <jats:sc>rban<\/jats:sc>\n                      C\n                      <jats:sc>lip<\/jats:sc>\n                      A\n                      <jats:sc>tlas<\/jats:sc>\n                    <\/jats:italic>\n                    supports scene retrieval through an augmented chat\u2010based interface and improves scene interpretation by tightly aligning textual outputs with video evidence. This design strengthens the connection between textual reasoning and visual evidence, reducing the effort required to validate model outputs and refine hypotheses. We demonstrate the usefulness of\n                    <jats:italic>\n                      U\n                      <jats:sc>rban<\/jats:sc>\n                      C\n                      <jats:sc>lip<\/jats:sc>\n                      A\n                      <jats:sc>tlas<\/jats:sc>\n                    <\/jats:italic>\n                    on the StreetAware dataset through two case studies involving hazardous scenarios and crossing dynamics at street intersections.\n                    <jats:italic>\n                      U\n                      <jats:sc>rban<\/jats:sc>\n                      C\n                      <jats:sc>lip<\/jats:sc>\n                      A\n                      <jats:sc>tlas<\/jats:sc>\n                    <\/jats:italic>\n                    helps analysts reason about safety\u2010 and mobility\u2010related patterns across large urban video collections.\n                  <\/jats:p>","DOI":"10.1111\/cgf.70431","type":"journal-article","created":{"date-parts":[[2026,6,2]],"date-time":"2026-06-02T08:48:07Z","timestamp":1780390087000},"update-policy":"https:\/\/doi.org\/10.1002\/crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["U\n                    <scp>rban<\/scp>\n                    C\n                    <scp>lip<\/scp>\n                    A\n                    <scp>tlas<\/scp>\n                    : A Visual Analytics Framework for Event and Scene Retrieval in Urban Videos"],"prefix":"10.1111","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-4917-4747","authenticated-orcid":false,"given":"Joel","family":"Perca","sequence":"first","affiliation":[{"name":"Funda\u00e7\u00e3o Getulio Vargas  Brazil"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-6547-7313","authenticated-orcid":false,"given":"Luis","family":"Sante","sequence":"additional","affiliation":[{"name":"Funda\u00e7\u00e3o Getulio Vargas  Brazil"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6126-4881","authenticated-orcid":false,"given":"Juanpablo","family":"Heredia","sequence":"additional","affiliation":[{"name":"Funda\u00e7\u00e3o Getulio Vargas  Brazil"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3341-7059","authenticated-orcid":false,"given":"Joao","family":"Rulff","sequence":"additional","affiliation":[{"name":"Funda\u00e7\u00e3o Getulio Vargas  Brazil"},{"name":"New York University  USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2452-2295","authenticated-orcid":false,"given":"Claudio","family":"Silva","sequence":"additional","affiliation":[{"name":"New York University  USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9096-6287","authenticated-orcid":false,"given":"Jorge","family":"Poco","sequence":"additional","affiliation":[{"name":"Funda\u00e7\u00e3o Getulio Vargas  Brazil"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"311","published-online":{"date-parts":[[2026,6,2]]},"reference":[{"key":"e_1_2_11_2_2","doi-asserted-by":"crossref","unstructured":"ArefeenM. A. DebnathB. ChakradharS.: TrafficLens: Multi-camera traffic video analysis using LLMs. InIEEE 27th International Conference on Intelligent Transportation Systems (ITSC)(Sept.2024) pp.3974\u20133981. doi:10.1109\/ITSC58415.2024.10920144. 3","DOI":"10.1109\/ITSC58415.2024.10920144"},{"key":"e_1_2_11_3_2","doi-asserted-by":"crossref","unstructured":"ArefeenM. A. DebnathB. Sarwar UddinM. Y. ChakradharS.: ViTA: An efficient video-to-text algorithm using VLM for RAG-based video analysis system. InIEEE\/CVF Conference on Computer Vision and Pattern Recognition Workshops (CVPRW)(June2024) pp.2266\u20132274. doi:10.1109\/CVPRW63382.2024.00232. 2","DOI":"10.1109\/CVPRW63382.2024.00232"},{"key":"e_1_2_11_4_2","doi-asserted-by":"crossref","unstructured":"ArefeenM. A. DebnathB. UddinM. Y. S. ChakradharS.:iRAG: Advancing RAG for videos with an incremental approach.4341\u20134348. arXiv:2404.12309. doi:10.1145\/3627673.3680088. 2","DOI":"10.1145\/3627673.3680088"},{"key":"e_1_2_11_5_2","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2024.3387972"},{"key":"e_1_2_11_6_2","doi-asserted-by":"publisher","DOI":"10.1088\/1757-899X\/671\/1\/012110"},{"key":"e_1_2_11_7_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW53098.2021.00475"},{"key":"e_1_2_11_8_2","doi-asserted-by":"crossref","unstructured":"BahmanyarR. HellekesJ. M\u00fchlhausM. GstaigerV. KurzF.: Traffic pattern analysis at urban intersections through vehicle detection in aerial imagery.ISPRS Annals of the Photogrammetry Remote Sensing and Spatial Information Sciences X-G-2025(2025) 151\u2013158. doi:10.5194\/isprs-annals-X-G-2025-151-2025. 2","DOI":"10.5194\/isprs-annals-X-G-2025-151-2025"},{"key":"e_1_2_11_9_2","volume-title":"The Economic and Societal Impact of Motor Vehicle Crashes, 2010 (Revised)","author":"Blincoe L. J.","year":"2015"},{"key":"e_1_2_11_10_2","doi-asserted-by":"crossref","unstructured":"ChenK. BanerjeeT. HuangX. RangarajanA. RankaS.: A visual analytics system for processed videos from traffic intersections. InProceedings of the 6th International Conference on Vehicle Technology and Intelligent Transport Systems (VEHITS 2020)(2020) BernsK. HelfertM. GusikhinO. (Eds.) SCITEPRESS pp.68\u201377. doi:10.5220\/0009422300680077. 3","DOI":"10.5220\/0009422300002550"},{"key":"e_1_2_11_11_2","doi-asserted-by":"crossref","unstructured":"CosciaA. EndertA.:VisPile: A visual analytics system for analyzing multiple text documents with large language models and knowledge graphs 2025. URL:https:\/\/arxiv.org\/abs\/2510.09605 arXiv:2510.09605. 7","DOI":"10.24251\/HICSS.2026.203"},{"key":"e_1_2_11_12_2","doi-asserted-by":"crossref","unstructured":"ChengB. MisraI. SchwingA. G. KirillovA. GirdharR.: Masked-attention mask transformer for universal image segmentation. In2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)(2022) pp.1280\u20131289. doi:10.1109\/CVPR52688.2022.00135. 5","DOI":"10.1109\/CVPR52688.2022.00135"},{"key":"e_1_2_11_13_2","doi-asserted-by":"publisher","DOI":"10.1109\/TVCG.2019.2934280"},{"key":"e_1_2_11_14_2","doi-asserted-by":"crossref","unstructured":"CaiW. PonomarenkoI. YuanJ. LiX. YangW. DongH. ZhaoB.: SpatialBot: Precise spatial understanding with vision language models. In2025 IEEE International Conference on Robotics and Automation (ICRA)(2025) pp.9490\u20139498. doi:10.1109\/ICRA55743.2025.11128671. 3","DOI":"10.1109\/ICRA55743.2025.11128671"},{"key":"e_1_2_11_15_2","unstructured":"ChengT. SongL. GeY. LiuW. WangX. ShanY.:YOLO-World: Real-time open-vocabulary object detection Feb.2024. arXiv:2401.17270. doi:10.48550\/arXiv.2401.17270. 3 5"},{"key":"e_1_2_11_16_2","doi-asserted-by":"publisher","DOI":"10.1155\/2017\/5202150"},{"key":"e_1_2_11_17_2","doi-asserted-by":"publisher","DOI":"10.1080\/13658816.2018.1510124"},{"key":"e_1_2_11_18_2","doi-asserted-by":"publisher","DOI":"10.1016\/j.cag.2025.104410"},{"key":"e_1_2_11_19_2","unstructured":"Federal Highway Administration:About intersection safety.https:\/\/highways.dot.gov\/safety\/intersection-safety\/about 2024. Last updated July 26 2024; accessed November 28 2025. 2"},{"key":"e_1_2_11_20_2","doi-asserted-by":"crossref","unstructured":"GuoZ. XiaL. YuY. AoT. HuangC.:LightRAG: Simple and fast retrieval-augmented generation Apr.2025. arXiv:2410.05779. doi:10.48550\/arXiv.2410.05779. 2","DOI":"10.18653\/v1\/2025.findings-emnlp.568"},{"key":"e_1_2_11_21_2","doi-asserted-by":"crossref","unstructured":"HerediaJ. Estrada-RaymeL. Matos-CangalayaJ. PocoJ.: Interactive exploration and explanation of spatio-temporal anomalies with graph-llm integration. In2025 38th SIBGRAPI Conference on Graphics Patterns and Images (SIBGRAPI)(2025) pp.1\u20136. doi:10.1109\/SIBGRAPI67909.2025.11223398. 2","DOI":"10.1109\/SIBGRAPI67909.2025.11223398"},{"key":"e_1_2_11_22_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i3.28032"},{"key":"e_1_2_11_23_2","unstructured":"JeongS. KimK. BaekJ. HwangS. J.:VideoRAG: Retrieval-augmented generation over video corpus May2025. arXiv:2501.05874. doi:10.48550\/arXiv.2501.05874. 2 3"},{"key":"e_1_2_11_24_2","unstructured":"LuoY. ZhengX. YangX. LiG. LinH. HuangJ. JiJ. ChaoF. LuoJ. JiR.:Video-RAG: Visually-aligned retrieval-augmented long video comprehension Dec.2024. arXiv:2411.13093. doi:10.48550\/arXiv.2411.13093. 3"},{"key":"e_1_2_11_25_2","unstructured":"MaoM. Perez-CabarcasM. M. KallakuriU. WaytowichN. R. LinX. MohseninT.:Multi-RAG: A multi-modal retrieval-augmented generation system for adaptive video understanding June2025. arXiv:2505.23990. doi:10.48550\/arXiv.2505.23990. 3"},{"key":"e_1_2_11_26_2","doi-asserted-by":"crossref","unstructured":"NunesA. L. DiazM. PocoJ.: MineTracker: Visual analytics for spatiotemporal analysis of mining areas in the brazilian amazon. In2025 38th SIBGRAPI Conference on Graphics Patterns and Images (SIBGRAPI)(2025) pp.1\u20136. doi:10.1109\/SIBGRAPI67909.2025.11223361. 3","DOI":"10.1109\/SIBGRAPI67909.2025.11223361"},{"key":"e_1_2_11_27_2","doi-asserted-by":"publisher","DOI":"10.3390\/s23073710"},{"key":"e_1_2_11_28_2","unstructured":"RulffJ. PereiraG. HosseiniM. LageM. SilvaC.:Towards data-informed interventions: Opportunities and challenges of street-level multimodal sensing 2024. URL:https:\/\/arxiv.org\/abs\/2410.22092 arXiv:2410.22092. 2"},{"key":"e_1_2_11_29_2","unstructured":"RenX. XuL. XiaL. WangS. YinD. HuangC.:VideoRAG: Retrieval-augmented generation with extreme long-context videos 2025. URL:https:\/\/arxiv.org\/abs\/2502.01549 arXiv:2502.01549. 2 3"},{"key":"e_1_2_11_30_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-03402-3_23"},{"key":"e_1_2_11_31_2","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2016.2568920"},{"key":"e_1_2_11_32_2","doi-asserted-by":"publisher","DOI":"10.1016\/j.aap.2019.105265"},{"key":"e_1_2_11_33_2","unstructured":"SankaradasM. RajendranR. K. ChakradharS. T.:StreamingRAG: Real-time contextual retrieval and generation framework Jan.2025. arXiv:2501.14101. doi:10.48550\/arXiv.2501.14101. 2"},{"key":"e_1_2_11_34_2","unstructured":"TanX. YeY. LuoY. WanQ. LiuF. CaiZ.:RAG-Adapter: A plug-and-play RAG-enhanced framework for long video understanding Mar.2025. arXiv:2503.08576. doi:10.48550\/arXiv.2503.08576. 3"},{"key":"e_1_2_11_35_2","unstructured":"WuT. GeS. QinJ. WuG. WangL.:Open-vocabulary spatio-temporal action detection 2024. URL:https:\/\/arxiv.org\/abs\/2405.10832 arXiv:2405.10832. 3"},{"key":"e_1_2_11_36_2","doi-asserted-by":"publisher","DOI":"10.1109\/TVCG.2018.2889081"}],"container-title":["Computer Graphics Forum"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/onlinelibrary.wiley.com\/doi\/pdf\/10.1111\/cgf.70431","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/onlinelibrary.wiley.com\/doi\/full-xml\/10.1111\/cgf.70431","content-type":"application\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/onlinelibrary.wiley.com\/doi\/pdf\/10.1111\/cgf.70431","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,2]],"date-time":"2026-06-02T08:48:25Z","timestamp":1780390105000},"score":1,"resource":{"primary":{"URL":"https:\/\/onlinelibrary.wiley.com\/doi\/10.1111\/cgf.70431"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6,2]]},"references-count":35,"alternative-id":["10.1111\/cgf.70431"],"URL":"https:\/\/doi.org\/10.1111\/cgf.70431","archive":["Portico"],"relation":{},"ISSN":["0167-7055","1467-8659"],"issn-type":[{"value":"0167-7055","type":"print"},{"value":"1467-8659","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,6,2]]},"assertion":[{"value":"2026-06-02","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}],"article-number":"e70431"}}