{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T04:24:18Z","timestamp":1750220658133,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":24,"publisher":"ACM","license":[{"start":{"date-parts":[[2020,12,7]],"date-time":"2020-12-07T00:00:00Z","timestamp":1607299200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2020,12,7]]},"DOI":"10.1145\/3429357.3430518","type":"proceedings-article","created":{"date-parts":[[2020,12,22]],"date-time":"2020-12-22T02:00:35Z","timestamp":1608602435000},"page":"1-7","source":"Crossref","is-referenced-by-count":3,"title":["mmFilter"],"prefix":"10.1145","author":[{"given":"Zhiming","family":"Hu","sequence":"first","affiliation":[{"name":"Samsung AI Center, Toronto, Canada"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ning","family":"Ye","sequence":"additional","affiliation":[{"name":"Samsung AI Center, Toronto, Canada"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Caleb","family":"Phillips","sequence":"additional","affiliation":[{"name":"Samsung AI Center, Toronto, Canada"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tim","family":"Capes","sequence":"additional","affiliation":[{"name":"Samsung AI Center, Toronto, Canada"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Iqbal","family":"Mohomed","sequence":"additional","affiliation":[{"name":"Samsung AI Center, Toronto, Canada"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2020,12,7]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"2020. Dense Convolutional Network (DenseNet). http:\/\/shorturl.at\/oJW58 Accessed: 2020-09-11.  2020. Dense Convolutional Network (DenseNet). http:\/\/shorturl.at\/oJW58 Accessed: 2020-09-11."},{"key":"e_1_3_2_1_2_1","unstructured":"2020. MobileNetV2 Feature Extrator. https:\/\/tfhub.dev\/google\/imagenet\/mobilenet_v2_100_224\/feature_vector\/4 Accessed: 2020-09-01.  2020. MobileNetV2 Feature Extrator. https:\/\/tfhub.dev\/google\/imagenet\/mobilenet_v2_100_224\/feature_vector\/4 Accessed: 2020-09-01."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"crossref","unstructured":"2020. Smart Home Security Cameras Market Size Share & Trends Analysis Report. http:\/\/shorturl.at\/grAY7 Accessed: 2020-09-11.  2020. Smart Home Security Cameras Market Size Share & Trends Analysis Report. http:\/\/shorturl.at\/grAY7 Accessed: 2020-09-11.","DOI":"10.1016\/S1353-4858(20)30129-X"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.572"},{"key":"e_1_3_2_1_5_1","unstructured":"Christopher Canel Thomas Kim Giulio Zhou Conglong Li Hyeontaek Lim David G Andersen Michael Kaminsky and Subramanya R Dulloor. 2019. Scaling Video Analytics on Constrained Edge Nodes. arXiv preprint arXiv:1905.13536 (2019).  Christopher Canel Thomas Kim Giulio Zhou Conglong Li Hyeontaek Lim David G Andersen Michael Kaminsky and Subramanya R Dulloor. 2019. Scaling Video Analytics on Constrained Edge Nodes. arXiv preprint arXiv:1905.13536 (2019)."},{"key":"e_1_3_2_1_6_1","first-page":"248","article-title":"Imagenet: A Large-Scale Hierarchical Image Database","author":"Deng Jia","year":"2009","journal-title":"Proc. of IEEE CVPR."},{"volume-title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. arXiv preprint arXiv:1810.04805","year":"2018","author":"Devlin Jacob","key":"e_1_3_2_1_7_1"},{"key":"e_1_3_2_1_8_1","unstructured":"Fartash Faghri David J Fleet Jamie Ryan Kiros and Sanja Fidler. 2017. VSE++: Improving Visual-Semantic Embeddings with Hard Negatives. arXiv preprint arXiv:1707.05612 (2017).  Fartash Faghri David J Fleet Jamie Ryan Kiros and Sanja Fidler. 2017. VSE++: Improving Visual-Semantic Embeddings with Hard Negatives. arXiv preprint arXiv:1707.05612 (2017)."},{"volume-title":"Rekall: Specifying Video Events using Compositions of Spatiotemporal Labels. arXiv preprint arXiv:1910.02993","year":"2019","author":"Fu Daniel Y","key":"e_1_3_2_1_9_1"},{"volume-title":"Squeeze-and-Excitation Networks. In Proceedings of the IEEE conference on computer vision and pattern recognition. 7132-7141","year":"2018","author":"Hu Jie","key":"e_1_3_2_1_10_1"},{"volume-title":"Noscope: Optimizing Neural Network Queries over Video at Scale. arXiv preprint arXiv:1703.02529","year":"2017","author":"Kang Daniel","key":"e_1_3_2_1_11_1"},{"volume-title":"Proc. of the IEEE\/CVF CVPR.","year":"2020","author":"Pishdad Mete","key":"e_1_3_2_1_12_1"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.83"},{"key":"e_1_3_2_1_14_1","first-page":"201","article-title":"Stacked Cross Attention for Image-Text Matching","author":"Lee Kuang-Huei","year":"2018","journal-title":"Proc. of ECCV."},{"key":"e_1_3_2_1_15_1","first-page":"359","article-title":"Reducto","author":"Li Yuanqi","year":"2020","journal-title":"On-Camera Filtering for Resource-Efficient Real-Time Video Analytics. In Proc. of ACM SIGCOMM."},{"key":"e_1_3_2_1_16_1","unstructured":"Y. Liu S. Albanie A. Nagrani and A. Zisserman. [n. d.]. Use What You Have: Video Retrieval using Representations from Collaborative Experts. In arXiv preprint arxiv:1907.13487 (2019).  Y. Liu S. Albanie A. Nagrani and A. Zisserman. [n. d.]. Use What You Have: Video Retrieval using Representations from Collaborative Experts. In arXiv preprint arxiv:1907.13487 (2019)."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.is.2013.10.006"},{"key":"e_1_3_2_1_18_1","unstructured":"Antoine Miech Ivan Laptev and Josef Sivic. 2017. Learnable Pooling with Context Gating for Video Classification. arXiv preprint arXiv:1706.06905 (2017).  Antoine Miech Ivan Laptev and Josef Sivic. 2017. Learnable Pooling with Context Gating for Video Classification. arXiv preprint arXiv:1706.06905 (2017)."},{"key":"e_1_3_2_1_19_1","unstructured":"Antoine Miech Ivan Laptev and Josef Sivic. 2018. Learning a Text-Video Embedding from Incomplete and Heterogeneous Data. arXiv:1804.02516 (2018).  Antoine Miech Ivan Laptev and Josef Sivic. 2018. Learning a Text-Video Embedding from Incomplete and Heterogeneous Data. arXiv:1804.02516 (2018)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/3081333.3089340"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00177"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00675"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.571"},{"volume-title":"Places: A 10 Million Image Database for Scene Recognition","year":"2017","author":"Zhou Bolei","key":"e_1_3_2_1_24_1"}],"event":{"name":"Middleware '20: 21st International Middleware Conference","sponsor":["ACM Association for Computing Machinery","IFIP"],"location":"Delft Netherlands","acronym":"Middleware '20"},"container-title":["Proceedings of the 1st International Middleware Conference Industrial Track"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3429357.3430518","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3429357.3430518","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T22:02:34Z","timestamp":1750197754000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3429357.3430518"}},"subtitle":["Language-Guided Video Analytics at the Edge"],"short-title":[],"issued":{"date-parts":[[2020,12,7]]},"references-count":24,"alternative-id":["10.1145\/3429357.3430518","10.1145\/3429357"],"URL":"https:\/\/doi.org\/10.1145\/3429357.3430518","relation":{},"subject":[],"published":{"date-parts":[[2020,12,7]]}}}