{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,4]],"date-time":"2026-05-04T10:15:07Z","timestamp":1777889707732,"version":"3.51.4"},"reference-count":66,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,10,19]],"date-time":"2025-10-19T00:00:00Z","timestamp":1760832000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,10,19]]},"DOI":"10.1109\/iccv51701.2025.00136","type":"proceedings-article","created":{"date-parts":[[2026,4,29]],"date-time":"2026-04-29T19:45:49Z","timestamp":1777491949000},"page":"1380-1390","source":"Crossref","is-referenced-by-count":0,"title":["BabyVLM: Data-Efficient Pretraining of VLMs Inspired by Infant Learning*"],"prefix":"10.1109","author":[{"given":"Shengao Wang Boston","family":"University","sequence":"first","affiliation":[{"name":"Boston University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Arjun","family":"Chandra","sequence":"additional","affiliation":[{"name":"Boston University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Aoming","family":"Liu","sequence":"additional","affiliation":[{"name":"Boston University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Venkatesh","family":"Saligrama","sequence":"additional","affiliation":[{"name":"Boston University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Boqing","family":"Gong","sequence":"additional","affiliation":[{"name":"Boston University"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.cmcl-1.4"},{"key":"ref2","volume-title":"Parikh: Visual Question Answering. In International Conference on Computer Vision (ICCV)","author":"Antol"},{"key":"ref3","article-title":"Qwen technical report","author":"Bai","year":"2023","journal-title":"arXiv preprint"},{"key":"ref4","article-title":"Qwen2. 5-vl technical report","author":"Bai","year":"2025","journal-title":"arXiv preprint"},{"key":"ref5","first-page":"65","article-title":"Meteor: An automatic metric for mt evaluation with improved correlation with human judgments","volume-title":"Proceedings of the acl workshop on intrinsic and extrinsic evaluation measures for machine translation and\/or summarization","author":"Banerjee"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1016\/j.cogpsych.2012.02.002"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.5040\/9781350934184.ch-004"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1016\/S0010-0277(96)00719-6"},{"key":"ref9","article-title":"[call for papers] the 2nd babylm challenge: Sample-efficient pretraining on a developmentally plausible corpus","author":"Choshen","year":"2024","journal-title":"arXiv preprint"},{"key":"ref10","article-title":"An image is worth $16 \\times 16$ words: Transformers for image recognition at scale","author":"Dosovitskiy","year":"2020","journal-title":"arXiv preprint"},{"key":"ref11","article-title":"Scaling rectified flow transformers for high-resolution image synthesis","volume-title":"Forty-first international conference on machine learning","author":"Esser"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.emnlp-main.154"},{"key":"ref13","article-title":"Analyzing and boosting the power of fine-grained visual recognition for multi-modal large language models","author":"He","year":"2025","journal-title":"arXiv preprint"},{"key":"ref14","article-title":"Training compute-optimal large language models","author":"Hoffmann","year":"2022","journal-title":"arXiv preprint"},{"key":"ref15","article-title":"Gpt-4o system card","author":"Hurst","year":"2024","journal-title":"arXiv preprint"},{"key":"ref16","author":"Jiang","year":"2023","journal-title":"Mistral 7b"},{"key":"ref17","first-page":"1514415169","article-title":"Mewl: Few-shot multimodal word learning with referential uncertainty","volume-title":"International Conference on Machine Learning","author":"Jiang"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-73778-7_164"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1037\/a0019165"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1038\/s41598-023-41538-y"},{"key":"ref21","author":"Li","year":"2024","journal-title":"Llava-next: What else influences visual instruction tuning beyond data?"},{"key":"ref22","author":"Li","year":"2024","journal-title":"Li-next: Stronger llms supercharge multimodal capabilities in the wild"},{"key":"ref23","article-title":"Llava-next-interleave: Tackling multi-image, video, and 3d in large multimodal models","author":"Li","year":"2024","journal-title":"arXiv preprint"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02484"},{"key":"ref26","author":"Liu","year":"2024","journal-title":"Llava-next: Improved reasoning, ocr, and world knowledge"},{"key":"ref27","article-title":"Visual instruction tuning","volume":"36","author":"Liu","year":"2024","journal-title":"Advances in neural information processing systems"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.32470\/vbbjtb0"},{"key":"ref29","article-title":"Deepseek-vl: towards real-world visionlanguage understanding","author":"Lu","year":"2024","journal-title":"arXiv preprint"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1017\/S0305000900013866"},{"key":"ref31","article-title":"Methods for assessing children\u2019s syntax","author":"McDaniel","year":"1998","journal-title":"Mit Press"},{"key":"ref32","article-title":"Im2text: Describing images using 1 million captioned photographs","volume":"24","author":"Ordonez","year":"2011","journal-title":"Advances in neural information processing systems"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1038\/s42256-024-00802-0"},{"key":"ref34","first-page":"9960","article-title":"Selfsupervised learning through the eyes of a child","volume":"33","author":"Orhan","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"key":"ref35","first-page":"409","article-title":"First language acquisition","author":"O\u2019Grady","year":"2001","journal-title":"Contemporary linguistics: An introduction"},{"key":"ref36","article-title":"A systematic investigation of learnability from single child linguistic input","author":"Qin","year":"2024","journal-title":"arXiv preprint"},{"issue":"8","key":"ref37","first-page":"9","article-title":"Language models are unsupervised multitask learners","volume":"1","author":"Radford","year":"2019","journal-title":"OpenAI blog"},{"key":"ref38","author":"Radford","year":"2021","journal-title":"Learning transferable visual models from natural language supervision"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1038\/s41598-024-72528-3"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1080\/15475440701360465"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.52202\/068431-1833"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P18-1238"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.52202\/075280-2360"},{"key":"ref44","doi-asserted-by":"crossref","DOI":"10.31234\/osf.io\/83gae","article-title":"Modelvsbaby: A developmentally motivated benchmark of out-of-distribution object recognition","volume-title":"PsyArXiv","author":"Sheybani","year":"2024"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.3389\/fpsyg.2017.02124"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1355"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.1162\/opmi_a_00039"},{"key":"ref48","article-title":"Devbench: A multimodal developmental benchmark for language learning","author":"Wei","year":"2024","journal-title":"arXiv preprint"},{"key":"ref49","article-title":"Gemini: a family of highly capable multimodal models","author":"Team","year":"2023","journal-title":"arXiv preprint"},{"key":"ref50","article-title":"Gemma: Open models based on gemini research and technology","author":"Team","year":"2024","journal-title":"arXiv preprint"},{"key":"ref51","article-title":"Clamp: contrastive language model prompt-tuning","author":"Teterwak","year":"2023","journal-title":"arXiv preprint"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr52688.2022.00517"},{"key":"ref53","article-title":"Llama: Open and efficient foundation language models","author":"Touvron","year":"2023","journal-title":"arXiv preprint"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1126\/science.adi1374"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1111\/cogs.13305"},{"key":"ref56","article-title":"Call for papers-the babylm challenge: Sample-efficient pretraining on a developmentally plausible corpus","author":"Warstadt","year":"2023","journal-title":"arXiv preprint"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.634"},{"key":"ref58","article-title":"Qwen2.5 technical report","author":"Yang","year":"2024","journal-title":"arXiv preprint"},{"key":"ref59","article-title":"The next big thing (s) in unsupervised machine learning: Five lessons from infant learning","author":"Zaadnoordijk","journal-title":"arXiv preprint"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.01100"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.542"},{"key":"ref62","author":"Zhang","year":"2024","journal-title":"Tinyllama: An open-source small language model"},{"key":"ref63","author":"Zhang","year":"2024","journal-title":"Llavanext: A strong zero-shot video understanding model"},{"key":"ref64","article-title":"Why are visually-grounded language models bad at image classification?","author":"Zhang","year":"2024","journal-title":"arXiv preprint"},{"key":"ref65","article-title":"Judging llm-as-a-judge with mt-bench and chatbot arena","author":"Zheng","year":"2023","journal-title":"arxiv. arXiv preprint"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1073\/pnas.2014196118"}],"event":{"name":"2025 IEEE\/CVF International Conference on Computer Vision (ICCV)","location":"Honolulu, HI, USA","start":{"date-parts":[[2025,10,19]]},"end":{"date-parts":[[2025,10,25]]}},"container-title":["2025 IEEE\/CVF International Conference on Computer Vision (ICCV)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11443115\/11443287\/11445796.pdf?arnumber=11445796","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,1]],"date-time":"2026-05-01T05:15:52Z","timestamp":1777612552000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11445796\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,19]]},"references-count":66,"URL":"https:\/\/doi.org\/10.1109\/iccv51701.2025.00136","relation":{},"subject":[],"published":{"date-parts":[[2025,10,19]]}}}