{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T15:57:42Z","timestamp":1783007862525,"version":"3.54.5"},"reference-count":58,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"4","license":[{"start":{"date-parts":[[2024,4,1]],"date-time":"2024-04-01T00:00:00Z","timestamp":1711929600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0\/"}],"funder":[{"name":"Dalio Philanthropies"},{"name":"Ocean X, Sea Grape Foundation"},{"name":"Virgin Unite"},{"name":"Rosamund Zander\/Hansjorg Wyss"},{"name":"Chris Anderson\/Jacqueline Novogratz"},{"name":"AI2050 Program","award":["G-22-63172"],"award-info":[{"award-number":["G-22-63172"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Robot. Autom. Lett."],"published-print":{"date-parts":[[2024,4]]},"DOI":"10.1109\/lra.2024.3366013","type":"journal-article","created":{"date-parts":[[2024,2,14]],"date-time":"2024-02-14T19:00:54Z","timestamp":1707937254000},"page":"3283-3290","source":"Crossref","is-referenced-by-count":24,"title":["Follow Anything: Open-Set Detection, Tracking, and Following in Real-Time"],"prefix":"10.1109","volume":"9","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-3800-9460","authenticated-orcid":false,"given":"Alaa","family":"Maalouf","sequence":"first","affiliation":[{"name":"Computer Science and Artificial Intelligence Lab, Massachusetts Institute of Technology, Cambridge, MA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1101-3336","authenticated-orcid":false,"given":"Ninad","family":"Jadhav","sequence":"additional","affiliation":[{"name":"John A. Paulson School Of Engineering And Applied Sciences, Harvard University, Boston, MA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4927-3387","authenticated-orcid":false,"given":"Krishna Murthy","family":"Jatavallabhula","sequence":"additional","affiliation":[{"name":"Computer Science and Artificial Intelligence Lab, Massachusetts Institute of Technology, Cambridge, MA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6807-7042","authenticated-orcid":false,"given":"Makram","family":"Chahine","sequence":"additional","affiliation":[{"name":"Computer Science and Artificial Intelligence Lab, Massachusetts Institute of Technology, Cambridge, MA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1286-7034","authenticated-orcid":false,"given":"Daniel M.","family":"Vogt","sequence":"additional","affiliation":[{"name":"John A. Paulson School Of Engineering And Applied Sciences, Harvard University, Boston, MA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7969-038X","authenticated-orcid":false,"given":"Robert J.","family":"Wood","sequence":"additional","affiliation":[{"name":"John A. Paulson School Of Engineering And Applied Sciences, Harvard University, Boston, MA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Antonio","family":"Torralba","sequence":"additional","affiliation":[{"name":"Computer Science and Artificial Intelligence Lab, Massachusetts Institute of Technology, Cambridge, MA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5473-3566","authenticated-orcid":false,"given":"Daniela","family":"Rus","sequence":"additional","affiliation":[{"name":"Computer Science and Artificial Intelligence Lab, Massachusetts Institute of Technology, Cambridge, MA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2019.2942944"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/icra48891.2023.10160827"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/INMIC.2013.6731341"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2018.2811762"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.4236\/wjet.2015.33C047"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2019.2953900"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00551"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58580-8_39"},{"key":"ref9","article-title":"Scalable multi-object identification for video object segmentation","author":"Yang","year":"2022"},{"key":"ref10","first-page":"2491","article-title":"Associating objects with transformers for video object segmentation","volume-title":"Proc. Int. Conf. Adv. Neural Inf. Process. Syst.","volume":"34","author":"Yang","year":"2021"},{"key":"ref11","first-page":"36324","article-title":"Decoupling features in hierarchical propagation for video object segmentation","volume-title":"Proc. Int. Conf. Adv. Neural Inf. Process. Syst.","volume":"35","author":"Yang","year":"2022"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/WACV56688.2023.00511"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00374"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00933"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00318"},{"key":"ref16","article-title":"Segment and track anything","author":"Cheng","year":"2023"},{"key":"ref17","first-page":"379","article-title":"R-FCN: Object detection via region-based fully convolutional networks","volume-title":"Proc. Int. Conf. Adv. Neural Inf. Process. Syst.","author":"Dai","year":"2016"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01079"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.81"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.169"},{"key":"ref21","first-page":"91","article-title":"Faster R-CNN: Towards real-time object detection with region proposal networks","volume-title":"Proc. Int. Conf. Adv. Neural Inf. Process. Syst.","volume":"28","author":"Ren","year":"2015"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.91"},{"key":"ref23","article-title":"YOLOv4: Optimal speed and accuracy of object detection","author":"Bochkovskiy","year":"2020"},{"key":"ref24","article-title":"YOLOv3: An incremental improvement","author":"Redmon","year":"2018"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.l007\/978-3-319-46448-0_2"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/TCSVT.2023.3255304"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/RCAR47638.2019.9043931"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/TITS.2022.3177627"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/ICUAS.2016.7502513"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1186\/s41074-019-0059-x"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/MICAI.2015.12"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/REDUAS47371.2019.8999675"},{"key":"ref33","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Radford","year":"2021"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"ref36","first-page":"4904","article-title":"Scaling up visual and vision-language representation learning with noisy text supervision","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Jia","year":"2021"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9747631"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1146\/annurev-control-101119-071628"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-main.703"},{"key":"ref40","first-page":"287","article-title":"Do as i can, not as i say: Grounding language in robotic affordances","volume-title":"Proc. 6th Conf. Robot Learn.","volume":"205","author":"Ahn","year":"2023"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.15607\/rss.2023.xix.025"},{"key":"ref42","first-page":"31199","article-title":"Pre-trained language models for interactive decision-making","volume-title":"Proc. Int. Conf. Adv. Neural Inf. Process. Syst.","volume":"35","author":"Li","year":"2022"},{"key":"ref43","first-page":"8821","article-title":"Zero-shot text-to-image generation","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Ramesh","year":"2021"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19836-6_6"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00209"},{"key":"ref46","article-title":"Hierarchical text-conditional image generation with clip latents","author":"Ramesh","year":"2022"},{"key":"ref47","first-page":"2022","article-title":"Decoupling features in hierarchical propagation for video object segmentation","volume-title":"Proc. Int. Conf. Adv. Neural Inf. Process. Syst.","author":"Yang"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00142"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00135"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-25069-9_3"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.15607\/rss.2023.xix.066"},{"key":"ref52","article-title":"Fast segment anything","author":"Zhao","year":"2023"},{"key":"ref53","article-title":"Grounding dino: Marrying dino with grounded pre-training for open-set object detection","author":"Liu","year":"2023"},{"key":"ref54","article-title":"Language-driven semantic segmentation","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Li","year":"2022"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.221"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20059-5_31"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01629"},{"key":"ref58","article-title":"CholecSeg8k: A semantic segmentation dataset for laparoscopic cholecystectomy based on cholec80","author":"Hong","year":"2020"}],"container-title":["IEEE Robotics and Automation Letters"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/7083369\/10440130\/10436161.pdf?arnumber=10436161","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,23]],"date-time":"2024-12-23T21:30:14Z","timestamp":1734989414000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10436161\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,4]]},"references-count":58,"journal-issue":{"issue":"4"},"URL":"https:\/\/doi.org\/10.1109\/lra.2024.3366013","relation":{},"ISSN":["2377-3766","2377-3774"],"issn-type":[{"value":"2377-3766","type":"electronic"},{"value":"2377-3774","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,4]]}}}