{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,18]],"date-time":"2025-12-18T14:23:59Z","timestamp":1766067839640,"version":"3.44.0"},"publisher-location":"New York, NY, USA","reference-count":16,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,12,7]],"date-time":"2023-12-07T00:00:00Z","timestamp":1701907200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,12,7]]},"DOI":"10.1145\/3628797.3628997","type":"proceedings-article","created":{"date-parts":[[2023,12,6]],"date-time":"2023-12-06T15:25:34Z","timestamp":1701876334000},"page":"960-965","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Vi-ATISO: An Effective Video Search Engine at AI Challenge HCMC 2023"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-7713-3184","authenticated-orcid":false,"given":"Quang-Tan","family":"Nguyen","sequence":"first","affiliation":[{"name":"University of Science, VNU-HCM, Viet Nam"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-3305-0971","authenticated-orcid":false,"given":"Xuan-Quang","family":"Nguyen","sequence":"additional","affiliation":[{"name":"University of Science, VNU-HCM, Viet Nam"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-1794-2814","authenticated-orcid":false,"given":"Trong-Bao","family":"Ho","sequence":"additional","affiliation":[{"name":"University of Science, VNU-HCM, Viet Nam"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-5167-1434","authenticated-orcid":false,"given":"Duc-Thang","family":"Truong","sequence":"additional","affiliation":[{"name":"University of Science, VNU-HCM, Viet Nam"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-5981-8055","authenticated-orcid":false,"given":"Minh-Hoang","family":"Le","sequence":"additional","affiliation":[{"name":"University of Science, VNU-HCM, Viet Nam"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,12,7]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-27077-2_48"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.1986.4767851"},{"key":"e_1_3_2_1_3_1","volume-title":"CLIP2Video: Mastering Video-Text Retrieval via Image CLIP. arXiv preprint arXiv:2106.11097","author":"Fang Han","year":"2021","unstructured":"Han Fang, Pengfei Xiong, Luhui Xu, and Yu Chen. 2021. CLIP2Video: Mastering Video-Text Retrieval via Image CLIP. arXiv preprint arXiv:2106.11097 (2021)."},{"key":"e_1_3_2_1_4_1","volume-title":"Video Event Retrieval with Flexible Textual-Visual Intermediary for VBS","author":"Hoang-Xuan Nhat","year":"2023","unstructured":"Nhat Hoang-Xuan, E-Ro Nguyen, Thang-Long Nguyen-Ho, Minh-Khoi Pham, Quang-Thuc Nguyen, Hoang-Phuc Trang-Trung, Van-Tu Ninh, Tu-Khiem Le, Cathal Gurrin, and Minh-Triet Tran. 2023. V-FIRST 2.0: Video Event Retrieval with Flexible Textual-Visual Intermediary for VBS 2023. In MultiMedia Modeling, Duc-Tien Dang-Nguyen, Cathal Gurrin, Martha Larson, Alan\u00a0F. Smeaton, Stevan Rudinac, Minh-Son Dao, Christoph Trattner, and Phoebe Chen (Eds.). Springer International Publishing, Cham, 652\u2013657."},{"key":"e_1_3_2_1_5_1","volume-title":"Microsoft COCO: Common Objects in Context. CoRR abs\/1405.0312","author":"Lin Tsung-Yi","year":"2014","unstructured":"Tsung-Yi Lin, Michael Maire, Serge\u00a0J. Belongie, Lubomir\u00a0D. Bourdev, Ross\u00a0B. Girshick, James Hays, Pietro Perona, Deva Ramanan, Piotr Doll\u00e1r, and C.\u00a0Lawrence Zitnick. 2014. Microsoft COCO: Common Objects in Context. CoRR abs\/1405.0312 (2014). arXiv:1405.0312http:\/\/arxiv.org\/abs\/1405.0312"},{"key":"e_1_3_2_1_6_1","volume-title":"ALADIN: Distilling Fine-grained Alignment Scores for Efficient Image-Text Matching and Retrieval. In International Conference on Content-based Multimedia Indexing. 64\u201370","author":"Messina Nicola","year":"2022","unstructured":"Nicola Messina, Matteo Stefanini, Marcella Cornia, Lorenzo Baraldi, Fabrizio Falchi, Giuseppe Amato, and Rita Cucchiara. 2022. ALADIN: Distilling Fine-grained Alignment Scores for Efficient Image-Text Matching and Retrieval. In International Conference on Content-based Multimedia Indexing. 64\u201370."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/219717.219748"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.findings-emnlp.92"},{"key":"e_1_3_2_1_9_1","volume-title":"Proceedings of the 37th International Conference on Machine Learning(ICML\u201920)","author":"Prokhorenkova Liudmila","year":"2020","unstructured":"Liudmila Prokhorenkova and Aleksandr Shekhovtsov. 2020. Graph-Based Nearest Neighbor Search: From Practice to Theory. In Proceedings of the 37th International Conference on Machine Learning(ICML\u201920). JMLR.org, Article 723, 11\u00a0pages."},{"key":"e_1_3_2_1_10_1","volume-title":"Learning Transferable Visual Models From Natural Language Supervision. CoRR abs\/2103.00020","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong\u00a0Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever. 2021. Learning Transferable Visual Models From Natural Language Supervision. CoRR abs\/2103.00020 (2021). arXiv:2103.00020https:\/\/arxiv.org\/abs\/2103.00020"},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"crossref","unstructured":"Jerome Revaud Jon Almazan Rafael\u00a0Sampaio de Rezende and Cesar\u00a0Roberto de Souza. 2019. Learning with Average Precision: Training Image Retrieval with a Listwise Loss. In ICCV.","DOI":"10.1109\/ICCV.2019.00521"},{"volume-title":"Exploring Effective Interactive Text-Based Video Search in vitrivr","author":"Sauter Loris","key":"e_1_3_2_1_12_1","unstructured":"Loris Sauter, Ralph Gasser, Silvan Heller, Luca Rossetto, Colin Saladin, Florian Spiess, and Heiko Schuldt. 2023. Exploring Effective Interactive Text-Based Video Search in vitrivr. In MultiMedia Modeling, Duc-Tien Dang-Nguyen, Cathal Gurrin, Martha Larson, Alan\u00a0F. Smeaton, Stevan Rudinac, Minh-Son Dao, Christoph Trattner, and Phoebe Chen (Eds.). Springer International Publishing, Cham, 646\u2013651."},{"key":"e_1_3_2_1_13_1","volume-title":"Vibro: Video Browsing with Semantic and Visual Image Embeddings","author":"Schall Konstantin","year":"2023","unstructured":"Konstantin Schall, Nico Hezel, Klaus Jung, and Kai\u00a0Uwe Barthel. 2023. Vibro: Video Browsing with Semantic and Visual Image Embeddings. In MultiMedia Modeling, Duc-Tien Dang-Nguyen, Cathal Gurrin, Martha Larson, Alan\u00a0F. Smeaton, Stevan Rudinac, Minh-Son Dao, Christoph Trattner, and Phoebe Chen (Eds.). Springer International Publishing, Cham, 665\u2013670."},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01838"},{"key":"e_1_3_2_1_15_1","volume-title":"Efficient and Accurate Arbitrary-Shaped Text Detection with Pixel Aggregation Network. CoRR abs\/1908.05900","author":"Wang Wenhai","year":"2019","unstructured":"Wenhai Wang, Enze Xie, Xiaoge Song, Yuhang Zang, Wenjia Wang, Tong Lu, Gang Yu, and Chunhua Shen. 2019. Efficient and Accurate Arbitrary-Shaped Text Detection with Pixel Aggregation Network. CoRR abs\/1908.05900 (2019). arXiv:1908.05900http:\/\/arxiv.org\/abs\/1908.05900"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"crossref","unstructured":"Haoyang Zhang Ying Wang Feras Dayoub and Niko S\u00fcnderhauf. 2021. VarifocalNet: An IoU-aware Dense Object Detector. In CVPR.","DOI":"10.1109\/CVPR46437.2021.00841"}],"event":{"name":"SOICT 2023: The 12th International Symposium on Information and Communication Technology","acronym":"SOICT 2023","location":"Ho Chi Minh Vietnam"},"container-title":["Proceedings of the 12th International Symposium on Information and Communication Technology"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3628797.3628997","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3628797.3628997","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T12:21:25Z","timestamp":1755778885000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3628797.3628997"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,12,7]]},"references-count":16,"alternative-id":["10.1145\/3628797.3628997","10.1145\/3628797"],"URL":"https:\/\/doi.org\/10.1145\/3628797.3628997","relation":{},"subject":[],"published":{"date-parts":[[2023,12,7]]},"assertion":[{"value":"2023-12-07","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}