{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,15]],"date-time":"2026-03-15T00:55:52Z","timestamp":1773536152941,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":22,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,3,16]]},"DOI":"10.1145\/3757279.3785641","type":"proceedings-article","created":{"date-parts":[[2026,3,10]],"date-time":"2026-03-10T00:27:38Z","timestamp":1773102458000},"page":"553-561","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["SCOPE: Real-Time Natural Language Camera Agent at the Edge"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-5472-2461","authenticated-orcid":false,"given":"Nikolaj","family":"Hindsbo","sequence":"first","affiliation":[{"name":"Armada, Bellevue, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6009-7612","authenticated-orcid":false,"given":"Sina","family":"Ehsani","sequence":"additional","affiliation":[{"name":"Armada, Bellevue, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6111-797X","authenticated-orcid":false,"given":"Pragyana","family":"Mishra","sequence":"additional","affiliation":[{"name":"Armada, Bellevue, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2026,3,16]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","unstructured":"Sandhini Agarwal Lama Ahmad Jason Ai Sam Altman Andy Applebaum Edwin Arbus Rahul K. Arora Yu Bai Bowen Baker Haiming Bao et al. 2025. gpt-oss-120b & gpt-oss-20b Model Card. arxiv:2508.10925. https:\/\/doi.org\/10.48550\/arXiv.2508.10925 10.48550\/arXiv.2508.10925","DOI":"10.48550\/arXiv.2508.10925"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","unstructured":"Michael Ahn Anthony Brohan Noah Brown Yevgen Chebotar Omar Cortes Byron David Chelsea Finn Chuyuan Fu Keerthana Gopalakrishnan Karol Hausman et al. 2022. Do as I can not as I say: Grounding language in robotic affordances. arXiv preprint arXiv:2204.01691 https:\/\/doi.org\/10.48550\/arXiv.2204.01691 10.48550\/arXiv.2204.01691","DOI":"10.48550\/arXiv.2204.01691"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00387"},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","unstructured":"Shuai Bai Keqin Chen Xuejing Liu Jialin Wang Wenbin Ge Sibo Song Kai Dang Peng Wang Shijie Wang Jun Tang et al. 2025. Qwen2.5-VL Technical Report. arXiv preprint arXiv:2502.13923 https:\/\/doi.org\/10.48550\/arXiv.2502.13923 10.48550\/arXiv.2502.13923","DOI":"10.48550\/arXiv.2502.13923"},{"key":"e_1_3_2_1_5_1","unstructured":"Lee Boonstra. 2024. Prompt Engineering. Google Cloud. https:\/\/www.kaggle.com\/whitepaper-prompt-engineering"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","unstructured":"Alexiy Buynitsky Sina Ehsani Bhanu Pallakonda and Pragyana Mishra. 2025. Camera Control at the Edge with Language Models for Scene Understanding. arXiv preprint arXiv:2505.06402 https:\/\/doi.org\/10.1109\/ICCAR64901.2025.11073044 10.1109\/ICCAR64901.2025.11073044","DOI":"10.1109\/ICCAR64901.2025.11073044"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.5555\/3692070.3692401"},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00008"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52734.2025.00018"},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","unstructured":"Dan Hendrycks Collin Burns Steven Basart Andy Zou Mantas Mazeika Dawn Song and Jacob Steinhardt. 2020. Measuring massive multitask language understanding. arXiv preprint arXiv:2009.03300 https:\/\/doi.org\/10.48550\/arXiv.2009.03300 10.48550\/arXiv.2009.03300","DOI":"10.48550\/arXiv.2009.03300"},{"key":"e_1_3_2_1_11_1","volume-title":"ART: Agent Reinforcement Trainer. https:\/\/github.com\/openpipe\/art","author":"Hilton Brad","year":"2025","unstructured":"Brad Hilton, Kyle Corbitt, David Corbitt, Saumya Gandhi, Angky William, Bohdan Kovalenskyi, and Andie Jones. 2025. ART: Agent Reinforcement Trainer. https:\/\/github.com\/openpipe\/art"},{"key":"e_1_3_2_1_12_1","unstructured":"Vikhyat Karamcheti. 2025. Moondream: Lightweight Vision\u2013Language Models (Moondream2 Moondream2-4bit Moondream3-preview). https:\/\/huggingface.co\/moondream Includes Moondream2 (2025-06-21) Moondream2-4bit (2025-04-14) Moondream3-preview (2025-09-18)"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","unstructured":"Jacky Liang Wenlong Huang Fei Xia Peng Xu Karol Hausman Brian Ichter Pete Florence and Andy Zeng. 2022. Code as policies: Language model programs for embodied control. arXiv preprint arXiv:2209.07753 https:\/\/doi.org\/10.1109\/ICRA48891.2023.10160591 10.1109\/ICRA48891.2023.10160591","DOI":"10.1109\/ICRA48891.2023.10160591"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00294"},{"key":"e_1_3_2_1_15_1","volume-title":"Proceedings of the 42nd International Conference on Machine Learning (Proceedings of Machine Learning Research","volume":"2025","author":"Patil Shishir G","unstructured":"Shishir G Patil, Huanzhi Mao, Fanjia Yan, Charlie Cheng-Jie Ji, Vishnu Suresh, Ion Stoica, and Joseph E. Gonzalez. 2025. The Berkeley Function Calling Leaderboard (BFCL): From Tool Use to Agentic Evaluation of Large Language Models. In Proceedings of the 42nd International Conference on Machine Learning (Proceedings of Machine Learning Research, Vol. 267). PMLR, Vancouver, Canada. https:\/\/openreview.net\/forum?id=2GmDdhBdDk ICML 2025 poster"},{"key":"e_1_3_2_1_16_1","unstructured":"Qwen Team. 2025. Qwen3 Technical Report. arxiv:2505.09388. https:\/\/doi.org\/10.48550\/arXiv.2505.09388 10.48550\/arXiv.2505.09388"},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.91"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","unstructured":"Nidhish Shah Zulkuf Genc and Dogu Araci. 2024. StackEval: Benchmarking LLMs in Coding Assistance. In Advances in Neural Information Processing Systems. 37 https:\/\/doi.org\/10.52202\/079017-1166 10.52202\/079017-1166","DOI":"10.52202\/079017-1166"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01075"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2024.3387941"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2401.16158"},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","unstructured":"Xiang Yue Tianyu Zheng Yuansheng Ni Yubo Wang Kai Zhang Shengbang Tong Yuxuan Sun Botao Yu Ge Zhang Huan Sun Yu Su Wenhu Chen and Graham Neubig. 2024. MMMU-Pro: A More Robust Multi-discipline Multimodal Understanding Benchmark. arXiv preprint arXiv:2409.02813 https:\/\/doi.org\/10.48550\/arXiv.2409.02813 10.48550\/arXiv.2409.02813","DOI":"10.48550\/arXiv.2409.02813"}],"event":{"name":"HRI '26: 21st ACM\/IEEE International Conference on Human-Robot Interaction","location":"Edinburgh Scotland UK","acronym":"HRI '26","sponsor":["SIGAI ACM Special Interest Group on Artificial Intelligence","SIGCHI ACM Special Interest Group on Computer-Human Interaction","IEEE RAS"]},"container-title":["Proceedings of the 21st ACM\/IEEE International Conference on Human-Robot Interaction"],"original-title":[],"deposited":{"date-parts":[[2026,3,15]],"date-time":"2026-03-15T00:37:36Z","timestamp":1773535056000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3757279.3785641"}},"subtitle":["A Sim-to-Real Benchmark and Analysis of Open-Source Vision and Language Agents for PTZ Camera Tasks"],"short-title":[],"issued":{"date-parts":[[2026,3,16]]},"references-count":22,"alternative-id":["10.1145\/3757279.3785641","10.1145\/3757279"],"URL":"https:\/\/doi.org\/10.1145\/3757279.3785641","relation":{},"subject":[],"published":{"date-parts":[[2026,3,16]]},"assertion":[{"value":"2026-03-16","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}