{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,22]],"date-time":"2026-04-22T18:18:03Z","timestamp":1776881883818,"version":"3.51.2"},"reference-count":43,"publisher":"IEEE","license":[{"start":{"date-parts":[[2022,5,23]],"date-time":"2022-05-23T00:00:00Z","timestamp":1653264000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2022,5,23]],"date-time":"2022-05-23T00:00:00Z","timestamp":1653264000000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022,5,23]]},"DOI":"10.1109\/icassp43922.2022.9747103","type":"proceedings-article","created":{"date-parts":[[2022,4,27]],"date-time":"2022-04-27T19:50:34Z","timestamp":1651089034000},"page":"7727-7731","source":"Crossref","is-referenced-by-count":19,"title":["Fast-Slow Transformer for Visually Grounding Speech"],"prefix":"10.1109","author":[{"given":"Puyuan","family":"Peng","sequence":"first","affiliation":[{"name":"The University of Texas at Austin,Department of Computer Science,Austin,Texas,USA,78712"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"David","family":"Harwath","sequence":"additional","affiliation":[{"name":"The University of Texas at Austin,Department of Computer Science,Austin,Texas,USA,78712"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1145\/3404835.3462829"},{"key":"ref38","article-title":"Bert: Pre-training of deep bidirectional transformers for language understanding","author":"devlin","year":"2019","journal-title":"NAACL"},{"key":"ref33","doi-asserted-by":"crossref","DOI":"10.1613\/jair.3994","article-title":"Framing image description as a ranking task: Data, models and evaluation metrics (extended abstract)","author":"hodosh","year":"2013","journal-title":"JAIR"},{"key":"ref32","article-title":"Microsoft coco: Common objects in context","author":"lin","year":"2014","journal-title":"ECCV"},{"key":"ref31","article-title":"Learning deep features for scene recognition using places database","author":"zhou","year":"2014","journal-title":"NeurIPS"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00970"},{"key":"ref37","article-title":"Gaussian error linear units (gelus)","author":"hendrycks","year":"2016"},{"key":"ref36","article-title":"Faster r-cnn: Towards real-time object detection with region proposal networks","author":"ren","year":"2015","journal-title":"TPAMI"},{"key":"ref35","article-title":"wav2vec 2.0: A framework for self-supervised learning of speech representations","author":"baevski","year":"2020","journal-title":"NeurIPS"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2016.2598339"},{"key":"ref10","article-title":"Learning hierarchical discrete linguistic units from visually-grounded speech","author":"harwath","year":"2020","journal-title":"ICLRE"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.findings-emnlp.244"},{"key":"ref11","article-title":"Jointly discovering visual objects and spoken words from raw sensory input","author":"harwath","year":"2019","journal-title":"IJCV"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414418"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-496"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682666"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.acl-long.411"},{"key":"ref16","article-title":"Vision as an inter-lingua: Learning multilingual semantic embeddings of untranscribed speech","author":"harwath","year":"2018","journal-title":"ICASSP"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.21437\/SLTU.2018-53"},{"key":"ref18","article-title":"Cat-playinginthesnow: Impact of prior segmentation on a model of visually grounded speech","author":"havard","year":"2020","journal-title":"CoNLL"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053428"},{"key":"ref28","article-title":"Zr-2021vg: Zero-resource speech challenge, visually-grounded language modelling track, 2021 edition","author":"alishahi","year":"2021"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1227"},{"key":"ref27","article-title":"The zero resource speech benchmark 2021: Metrics and baselines for unsupervised spoken language modeling","author":"nguyen","year":"2020","journal-title":"SAS NeurIPS"},{"key":"ref3","article-title":"End-to-end multi-modal speech recognition","author":"palaskar","year":"2018","journal-title":"ICASSP"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-3067"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1514"},{"key":"ref5","doi-asserted-by":"crossref","DOI":"10.21437\/Interspeech.2017-502","article-title":"Visually grounded learning of keyword prediction from untran-scribed speech","author":"kamper","year":"2017","journal-title":"InterSpeech"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-435"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1148"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2016.7846320"},{"key":"ref9","article-title":"Representations of language in a model of visually grounded speech signal","author":"chrupa?a","year":"2017","journal-title":"ACL"},{"key":"ref1","article-title":"Visually grounded models of spoken language: A survey of datasets, architectures and evaluation techniques","author":"chrupa?a","year":"2021"},{"key":"ref20","article-title":"Symbolic inductive bias for visually grounded learning of spoken language","author":"chrupa?a","year":"2019","journal-title":"ACL"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/K19-1006"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-3024"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-1182"},{"key":"ref24","article-title":"Attention is all you need","author":"vaswani","year":"2017","journal-title":"NeurIPS"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"ref23","article-title":"Talk, don&#x2019;t write: A study of direct speech-based image retrieval","author":"sanabria","year":"2021","journal-title":"INTER-SPEECH"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2015.7404800"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-1465"},{"key":"ref25","article-title":"Unsupervised learning of spoken language with visual context","author":"harwath","year":"2016","journal-title":"NIPS"}],"event":{"name":"ICASSP 2022 - 2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","location":"Singapore, Singapore","start":{"date-parts":[[2022,5,23]]},"end":{"date-parts":[[2022,5,27]]}},"container-title":["ICASSP 2022 - 2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9745891\/9746004\/09747103.pdf?arnumber=9747103","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,8,15]],"date-time":"2022-08-15T20:08:46Z","timestamp":1660594126000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9747103\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,5,23]]},"references-count":43,"URL":"https:\/\/doi.org\/10.1109\/icassp43922.2022.9747103","relation":{},"subject":[],"published":{"date-parts":[[2022,5,23]]}}}