{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,25]],"date-time":"2026-07-25T00:40:25Z","timestamp":1784940025343,"version":"3.55.0"},"reference-count":85,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2024,10,15]],"date-time":"2024-10-15T00:00:00Z","timestamp":1728950400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,15]],"date-time":"2024-10-15T00:00:00Z","timestamp":1728950400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"Major Project of Anhui Province","award":["202203a05020011"],"award-info":[{"award-number":["202203a05020011"]}]},{"DOI":"10.13039\/501100012166","name":"National Key R&D Program of China","doi-asserted-by":"crossref","award":["2022YFB4500601"],"award-info":[{"award-number":["2022YFB4500601"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["72188101, 62272144, 62020106007, U20A20183"],"award-info":[{"award-number":["72188101, 62272144, 62020106007, U20A20183"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Comput Vis"],"published-print":{"date-parts":[[2025,4]]},"DOI":"10.1007\/s11263-024-02261-x","type":"journal-article","created":{"date-parts":[[2024,10,15]],"date-time":"2024-10-15T02:01:53Z","timestamp":1728957713000},"page":"1644-1664","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":48,"title":["Audio-Visual Segmentation with Semantics"],"prefix":"10.1007","volume":"133","author":[{"given":"Jinxing","family":"Zhou","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xuyang","family":"Shen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jianyuan","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jiayi","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Weixuan","family":"Sun","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jing","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Stan","family":"Birchfield","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dan","family":"Guo","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Lingpeng","family":"Kong","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Meng","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1404-3610","authenticated-orcid":false,"given":"Yiran","family":"Zhong","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,10,15]]},"reference":[{"key":"2261_CR1","doi-asserted-by":"crossref","unstructured":"Afouras, T., Owens, A., Chung, J.S., & Zisserman, A. (2020). Self-supervised learning of audio-visual objects from video. In: proceedings of the european conference on computer vision (ECCV), pp. 208\u2013224.","DOI":"10.1007\/978-3-030-58523-5_13"},{"key":"2261_CR2","doi-asserted-by":"crossref","unstructured":"Arandjelovic, R., & Zisserman, A. (2017). Look, listen and learn. In: proceedings of the IEEE international conference on computer vision (ICCV), pp. 609\u2013617","DOI":"10.1109\/ICCV.2017.73"},{"key":"2261_CR3","doi-asserted-by":"crossref","unstructured":"Arandjelovic, R., & Zisserman, A. (2018). Objects that sound. In: proceedings of the European conference on computer vision (ECCV), pp. 435\u2013451.","DOI":"10.1007\/978-3-030-01246-5_27"},{"key":"2261_CR4","doi-asserted-by":"crossref","unstructured":"Aytar, Y., Vondrick, C., Torralba, A. (2016). Soundnet: Learning sound representations from unlabeled video. Advances in Neural Information Processing Systems (NeurIPS).","DOI":"10.1109\/CVPR.2016.18"},{"key":"2261_CR5","doi-asserted-by":"crossref","unstructured":"Botach, A., Zheltonozhskii, E., Baskin, C. (2022). End-to-end referring video object segmentation with multimodal transformers. In: proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp. 4985\u20134995.","DOI":"10.1109\/CVPR52688.2022.00493"},{"key":"2261_CR6","doi-asserted-by":"crossref","unstructured":"Caelles, S., Maninis, K.K., Pont-Tuset, J., Leal-Taix\u00e9, L., Cremers, D., & Van\u00a0Gool, L. (2017). One-shot video object segmentation. In: Proceedings of the IEEE conference on computer vision and pattern recognition (CVPR), pp. 221\u2013230.","DOI":"10.1109\/CVPR.2017.565"},{"key":"2261_CR7","doi-asserted-by":"crossref","unstructured":"Chen, H., Xie, W., Vedaldi, A., & Zisserman, A. (2020). VGGSound: A large-scale audio-visual dataset. In: IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 721\u2013725.","DOI":"10.1109\/ICASSP40776.2020.9053174"},{"key":"2261_CR8","unstructured":"Chen, H., Xie, W., Afouras, T., Nagrani, A., Vedaldi, A., & Zisserman, A. (2021a). Audio-visual synchronisation in the wild. In: british machine vision conference (BMVC), pp. 1\u201324."},{"key":"2261_CR9","doi-asserted-by":"crossref","unstructured":"Chen, H., Xie, W., Afouras, T., Nagrani, A., Vedaldi, A., & Zisserman, A. (2021b). Localizing visual sounds the hard way. In: proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp. 16867\u201316876.","DOI":"10.1109\/CVPR46437.2021.01659"},{"key":"2261_CR10","doi-asserted-by":"crossref","unstructured":"Chen, L.C., Papandreou, G., Kokkinos, I., Murphy, K., & Yuille, A.L. (2017). DeepLab: Semantic image segmentation with deep convolutional nets, atrous convolution, and fully connected crfs. IEEE transactions on pattern analysis and machine intelligence (TPAMI) pp. 834\u2013848.","DOI":"10.1109\/TPAMI.2017.2699184"},{"key":"2261_CR11","doi-asserted-by":"crossref","unstructured":"Chen, Y., Pont-Tuset, J., Montes, A., & Van\u00a0Gool, L. (2018). Blazingly fast video object segmentation with pixel-wise metric learning. In: Proceedings of the IEEE Conference on computer vision and pattern recognition (CVPR), pp. 1189\u20131198.","DOI":"10.1109\/CVPR.2018.00130"},{"key":"2261_CR12","doi-asserted-by":"crossref","unstructured":"Cheng, H., Liu, Z., Zhou, H., Qian, C., Wu, W., Wang, L. (2022). Joint-modal label denoising for weakly-supervised audio-visual video parsing. proceedings of the European conference on computer vision (ECCV) pp. 431\u2013448.","DOI":"10.1007\/978-3-031-19830-4_25"},{"key":"2261_CR13","unstructured":"Cheng, J., Liu, S., Tsai, Y.H., Hung, W.C., De\u00a0Mello, S., Gu, J., Kautz, J., Wang, S., & Yang, M.H. (2017). Learning to segment instances in videos with spatial propagation network. arXiv preprint arXiv:1709.04609."},{"key":"2261_CR14","doi-asserted-by":"crossref","unstructured":"Cheng, Y., Wang, R., Pan, Z., Feng, R., & Zhang, Y. (2020). Look, listen, and attend: Co-attention network for self-supervised audio-visual representation learning. In: proceedings of the 28th ACM international conference on multimedia (ACM MM), pp. 3884\u20133892.","DOI":"10.1145\/3394171.3413869"},{"key":"2261_CR15","doi-asserted-by":"crossref","unstructured":"Chung, J.S., & Zisserman, A. (2017). Lip reading in the wild. In: Asian conference on computer vision (ACCV), pp. 87\u2013103.","DOI":"10.1007\/978-3-319-54184-6_6"},{"key":"2261_CR16","doi-asserted-by":"crossref","unstructured":"Duke, B., Ahmed, A., Wolf, C., Aarabi, P., & Taylor, G.W. (2021). SSTVOS: Sparse spatiotemporal transformers for video object segmentation. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp. 5912\u20135921.","DOI":"10.1109\/CVPR46437.2021.00585"},{"key":"2261_CR17","doi-asserted-by":"crossref","unstructured":"Faktor, A., & Irani, M. (2014). Video segmentation by non-local consensus voting. In: British Machine Vision Conference (BMVC), pp. 1\u20138.","DOI":"10.5244\/C.28.21"},{"key":"2261_CR18","doi-asserted-by":"crossref","unstructured":"Gao, R., & Grauman, K. (2019). Co-separating sounds of visual objects. In: proceedings of the IEEE\/CVF international conference on computer vision (ICCV), pp. 3879\u20133888.","DOI":"10.1109\/ICCV.2019.00398"},{"key":"2261_CR19","doi-asserted-by":"crossref","unstructured":"Gao, R., Feris, R., & Grauman, K. (2018). Learning to separate object sounds by watching unlabeled video. In: proceedings of the European conference on computer vision (ECCV), pp. 35\u201353.","DOI":"10.1007\/978-3-030-01219-9_3"},{"key":"2261_CR20","doi-asserted-by":"crossref","unstructured":"Gemmeke JF, Ellis DP, Freedman, D., Jansen, A., Lawrence, W., Moore, R.C., Plakal, M., & Ritter, M. (2017). Audio set: An ontology and human-labeled dataset for audio events. In: IEEE international conference on acoustics, speech and signal processing (ICASSP), IEEE, pp. 776\u2013780.","DOI":"10.1109\/ICASSP.2017.7952261"},{"key":"2261_CR21","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., & Sun, J. (2016). Deep residual learning for image recognition. In: proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp. 770\u2013778.","DOI":"10.1109\/CVPR.2016.90"},{"key":"2261_CR22","doi-asserted-by":"crossref","unstructured":"Hershey, S., Chaudhuri, S., Ellis, D.P., Gemmeke, J.F., Jansen, A., Moore, R.C., Plakal, M., Platt, D., Saurous, R.A., Seybold, B., et\u00a0al. (2017). CNN architectures for large-scale audio classification. In: IEEE international conference on acoustics, speech and signal processing (ICASSP), pp. 131\u2013135.","DOI":"10.1109\/ICASSP.2017.7952132"},{"key":"2261_CR23","doi-asserted-by":"crossref","unstructured":"Hu, D., Nie, F., & Li, X. (2019). Deep multimodal clustering for unsupervised audiovisual learning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp. 9248\u20139257.","DOI":"10.1109\/CVPR.2019.00947"},{"key":"2261_CR24","first-page":"10077","volume":"33","author":"D Hu","year":"2020","unstructured":"Hu, D., Qian, R., Jiang, M., Tan, X., Wen, S., Ding, E., Lin, W., & Dou, D. (2020). Discriminative sounding objects localization via self-supervised audiovisual matching. Adv. Neural Inform. Process. Syst., 33, 10077\u201310087.","journal-title":"Adv. Neural Inform. Process. Syst."},{"key":"2261_CR25","doi-asserted-by":"crossref","unstructured":"Hu, X., Chen, Z., Owens, A. (2022). Mix and localize: Localizing sound sources in mixtures. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp. 10483\u201310492.","DOI":"10.1109\/CVPR52688.2022.01023"},{"key":"2261_CR26","first-page":"20","volume":"33","author":"YT Hu","year":"2017","unstructured":"Hu, Y. T., Huang, J. B., & Schwing, A. (2017). Maskrnn: Instance level video object segmentation. Adv. Neural Inform. Process. Syst., 33, 20.","journal-title":"Adv. Neural Inform. Process. Syst."},{"key":"2261_CR27","doi-asserted-by":"crossref","unstructured":"Jiang, X., Xu, X., Chen, Z., Zhang, J., Song, J., Shen, F., Lu, H., Shen, H.T. (2022). Dhhn: Dual hierarchical hybrid network for weakly-supervised audio-visual video parsing. In: Proceedings of the 30th ACM international conference on multimedia (ACM MM), pp. 719\u2013727.","DOI":"10.1145\/3503161.3548309"},{"key":"2261_CR28","doi-asserted-by":"crossref","unstructured":"Khoreva, A., Rohrbach, A., & Schiele, B. (2018). Video object segmentation with language referring expressions. In: proceedings of the Asian conference on computer vision (ACCV), pp. 123\u2013141.","DOI":"10.1007\/978-3-030-20870-7_8"},{"key":"2261_CR29","doi-asserted-by":"crossref","unstructured":"Kirillov, A., Girshick, R., He, K., & Doll\u00e1r, P. (2019). Panoptic feature pyramid networks. In: proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp. 6399\u20136408.","DOI":"10.1109\/CVPR.2019.00656"},{"key":"2261_CR30","doi-asserted-by":"crossref","unstructured":"Kirillov, A., Mintun, E., Ravi, N., Mao, H., Rolland, C., Gustafson, L., Xiao, T., Whitehead, S., Berg, A.C., Lo, W.Y., et\u00a0al. (2023). Segment anything. arXiv preprint arXiv:2304.02643.","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"2261_CR31","doi-asserted-by":"crossref","unstructured":"Lin, Y.B., & Wang, Y.C.F. (2020). Audiovisual transformer with instance attention for audio-visual event localization. In: proceedings of the Asian conference on computer vision (ACCV).","DOI":"10.1007\/978-3-030-69544-6_17"},{"key":"2261_CR32","doi-asserted-by":"crossref","unstructured":"Lin, Y. B., Li, Y. J., & Wang, Y. C. F. (2019). Dual-modality seq2seq network for audio-visual event localization. IEEE international conference on acoustics (pp. 2002\u20132006). Speech and Signal Processing (ICASSP): IEEE.","DOI":"10.1109\/ICASSP.2019.8683226"},{"key":"2261_CR33","unstructured":"Lin, Y.B., Tseng, H.Y., Lee, H.Y., Lin, Y.Y., & Yang, M.H. (2021). Exploring cross-video and cross-modality signals for weakly-supervised audio-visual video parsing. Advances in Neural Information Processing Systems (NeurIPS)."},{"key":"2261_CR34","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lin, Y., Cao, Y., Hu, H., Wei, Y., Zhang, Z., Lin, S., Guo, B. (2021). Swin transformer: Hierarchical vision transformer using shifted windows. In: proceedings of the IEEE\/CVF international conference on computer vision (ICCV), pp. 10012\u201310022.","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"2261_CR35","unstructured":"bibitem[Long et al.Long et al.2015]jon2014fcn Long, J., Shelhamer, E., & Darrell, T. (2015). Fully convolutional networks for semantic segmentation. In: pProceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp. 3431\u20133440."},{"key":"2261_CR36","unstructured":"Van\u00a0der Maaten, L., & Hinton, G. (2008). Visualizing data using t-sne. J. Mach. Learn. Res. (JMLR)."},{"key":"2261_CR37","unstructured":"Mahadevan, S., Athar, A., O\u0161ep, A., Hennen, S., Leal-Taix\u00e9, L., & Leibe, B. (2020). Making a case for 3D convolutions for object segmentation in videos. In: British machine vision conference (BMVC), pp. 1\u201315."},{"key":"2261_CR38","doi-asserted-by":"crossref","unstructured":"Mahmud, T., Tian, Y., & Marculescu, D. (2024). T-vsl: Text-guided visual sound source localization in mixtures. In: proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp. 26742\u201326751.","DOI":"10.1109\/CVPR52733.2024.02525"},{"key":"2261_CR39","unstructured":"Mao, Y., Zhang, J., Wan, Z., Dai, Y., Li, A., Lv, Y., Tian, X., Fan, D.P., Barnes, N. (2021). Transformer transforms salient object detection and camouflaged object detection. arXiv preprint arXiv:2104.10127."},{"key":"2261_CR40","doi-asserted-by":"crossref","unstructured":"Mo, S., Morgado, P. (2022). Localizing visual sounds the easy way. In: European conference on computer vision (ECCV), pp. 218\u2013234.","DOI":"10.1007\/978-3-031-19836-6_13"},{"key":"2261_CR41","unstructured":"Mo, S., & Morgado, P. (2023). A unified audio-visual learning framework for localization, separation, and recognition. In: international conference on machine learning (ICML), pp. 25006\u201325017."},{"key":"2261_CR42","doi-asserted-by":"crossref","unstructured":"Mo, S., & Tian, Y. (2023). Audio-visual grouping network for sound localization from mixtures. In: proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp. 10565\u201310574.","DOI":"10.1109\/CVPR52729.2023.01018"},{"key":"2261_CR43","doi-asserted-by":"crossref","unstructured":"Owens, A., & Efros, A.A. (2018). Audio-visual scene analysis with self-supervised multisensory features. In: proceedings of the european conference on computer vision (ECCV), pp. 631\u2013648.","DOI":"10.1007\/978-3-030-01231-1_39"},{"key":"2261_CR44","doi-asserted-by":"crossref","unstructured":"Park, S., Senocak, A., & Chung, J.S. (2024). Can clip help sound source localization? In: proceedings of the IEEE\/CVF winter conference on applications of computer vision (WACV), pp. 5711\u20135720.","DOI":"10.1109\/WACV57701.2024.00561"},{"key":"2261_CR45","first-page":"19","volume":"32","author":"A Paszke","year":"2019","unstructured":"Paszke, A., Gross, S., Massa, F., Lerer, A., Bradbury, J., Chanan, G., Killeen, T., Lin, Z., et al. (2019). Pytorch: an imperative style, high-performance deep learning library. Adv. Neural Inform. Process. Syst., 32, 19.","journal-title":"Adv. Neural Inform. Process. Syst."},{"key":"2261_CR46","doi-asserted-by":"crossref","unstructured":"Perazzi, F., Pont-Tuset, J., McWilliams, B., Van\u00a0Gool, L., Gross, M., Sorkine-Hornung, A. (2016). A benchmark dataset and evaluation methodology for video object segmentation. In: proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp. 724\u2013732.","DOI":"10.1109\/CVPR.2016.85"},{"key":"2261_CR47","doi-asserted-by":"crossref","unstructured":"Qian, R., Hu, D., Dinkel, H., Wu, M., Xu, N., Lin, W. (2020). Multiple sound sources localization from coarse to fine. In: Proceedings of the European conference on computer vision (ECCV), pp. 292\u2013308.","DOI":"10.1007\/978-3-030-58565-5_18"},{"key":"2261_CR48","doi-asserted-by":"crossref","unstructured":"Ramaswamy, J. (2020). What makes the sound?: A dual-modality interacting network for audio-visual event localization. IEEE International Conference on Acoustics (pp. 4372\u20134376). Speech and Signal Processing (ICASSP): IEEE.","DOI":"10.1109\/ICASSP40776.2020.9053895"},{"key":"2261_CR49","doi-asserted-by":"crossref","unstructured":"Ramaswamy, J., & Das, S. (2020). See the sound, hear the pixels. In: proceedings of the IEEE\/CVF winter conference on applications of computer vision (WACV), pp. 2970\u20132979.","DOI":"10.1109\/WACV45572.2020.9093616"},{"key":"2261_CR50","doi-asserted-by":"crossref","unstructured":"Ronneberger, O., Fischer, P., & Brox, T. (2015). U-net: Convolutional networks for biomedical image segmentation. In: international conference on medical image computing and computer-assisted intervention (MICCAI), pp. 234\u2013241.","DOI":"10.1007\/978-3-319-24574-4_28"},{"key":"2261_CR51","doi-asserted-by":"crossref","unstructured":"Rouditchenko, A., Zhao, H., Gan, C., McDermott, J., & Torralba, A. (2019). Self-supervised audio-visual co-segmentation. IEEE international conference on acoustics (pp. 2357\u20132361). Speech and Signal Processing (ICASSP): IEEE.","DOI":"10.1109\/ICASSP.2019.8682467"},{"key":"2261_CR52","doi-asserted-by":"crossref","unstructured":"Russakovsky, O., Deng, J., Su, H., Krause, J., Satheesh, S., Ma, S., Huang, Z., Karpathy, A., Khosla, A., Bernstein, M., et\u00a0al. (2015). ImageNet large scale visual recognition challenge. International Journal of Computer Vision (IJCV) pp. 211\u2013252.","DOI":"10.1007\/s11263-015-0816-y"},{"key":"2261_CR53","doi-asserted-by":"crossref","unstructured":"Selvaraju, R.R., Cogswell, M., Das, A., Vedantam, R., Parikh, D., Batra, D. (2017). Grad-cam: Visual explanations from deep networks via gradient-based localization. In: proceedings of the IEEE international conference on computer vision (ICCV), pp. 618\u2013626.","DOI":"10.1109\/ICCV.2017.74"},{"key":"2261_CR54","doi-asserted-by":"crossref","unstructured":"Senocak, A., Oh, T.H., Kim, J., Yang, M.H., Kweon, I.S. (2018). Learning to localize sound source in visual scenes. In: proceedings of the IEEE conference on computer vision and pattern recognition (CVPR), pp. 4358\u20134366.","DOI":"10.1109\/CVPR.2018.00458"},{"key":"2261_CR55","doi-asserted-by":"crossref","unstructured":"Seo, S., Lee, J.Y., & Han, B. (2020). Urvos: Unified referring video object segmentation network with a large-scale benchmark. In: proceedings of the european conference on computer vision (ECCV), pp. 208\u2013223.","DOI":"10.1007\/978-3-030-58555-6_13"},{"key":"2261_CR56","doi-asserted-by":"crossref","unstructured":"Son\u00a0Chung, J., Senior, A., Vinyals, O., Zisserman, A. (2017). Lip reading sentences in the wild. In: proceedings of the IEEE conference on computer vision and pattern recognition (CVPR), pp. 6447\u20136456.","DOI":"10.1109\/CVPR.2017.367"},{"key":"2261_CR57","doi-asserted-by":"crossref","unstructured":"Song, H., Wang, W., Zhao, S., Shen, J., & Lam, K.M. (2018). Pyramid dilated deeper convlstm for video salient object detection. In: Proceedings of the European conference on computer vision (ECCV), pp. 715\u2013731.","DOI":"10.1007\/978-3-030-01252-6_44"},{"key":"2261_CR58","doi-asserted-by":"crossref","unstructured":"Tian, Y., Shi, J., Li, B., Duan, Z., & Xu, C. (2018). Audio-visual event localization in unconstrained videos. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 247\u2013263.","DOI":"10.1007\/978-3-030-01216-8_16"},{"key":"2261_CR59","doi-asserted-by":"crossref","unstructured":"Tian, Y., Li, D., & Xu, C. (2020). Unified multisensory perception: Weakly-supervised audio-visual video parsing. In: proceedings of the European conference on computer vision (ECCV), pp. 436\u2013454.","DOI":"10.1007\/978-3-030-58580-8_26"},{"key":"2261_CR60","doi-asserted-by":"crossref","unstructured":"Tokmakov, P., Alahari, K., & Schmid, C. (2017). Learning motion patterns in videos. In: Proceedings of the IEEE conference on computer vision and pattern recognition (CVPR), pp. 3386\u20133394.","DOI":"10.1109\/CVPR.2017.64"},{"key":"2261_CR61","doi-asserted-by":"crossref","unstructured":"Ventura, C., Bellver, M., Girbau, A., Salvador, A., Marques, F., Giro-i Nieto, X. (2019). Rvos: End-to-end recurrent network for video object segmentation. In: proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp. 5277\u20135286.","DOI":"10.1109\/CVPR.2019.00542"},{"key":"2261_CR62","unstructured":"Vijayanarasimhan, S., Ricco, S., Schmid, C., Sukthankar, R., Fragkiadaki, K. (2017). Sfm-net: Learning of structure and motion from video. arXiv preprint arXiv:1704.07804."},{"key":"2261_CR63","unstructured":"Wang, H., Zha, Z.J., Li, L., Chen, X., & Luo, J. (2022a). Semantic and relation modulation for audio-visual event localization. IEEE Trans. Patt. Anal. Mach. Intell. (TPAMI) pp. 1\u201315."},{"key":"2261_CR64","doi-asserted-by":"crossref","unstructured":"Wang, W., Xie, E., Li, X., Fan, D.P., Song, K., Liang, D., Lu, T., Luo, P., & Shao, L. (2022b). PVTv2: Improved baselines with pyramid vision transformer. Computat. Visual Media pp. 1\u201310.","DOI":"10.1007\/s41095-022-0274-8"},{"key":"2261_CR65","doi-asserted-by":"crossref","unstructured":"Wang, X., Girshick, R., Gupta, A., & He, K. (2018). Non-local neural networks. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp. 7794\u20137803.","DOI":"10.1109\/CVPR.2018.00813"},{"key":"2261_CR66","doi-asserted-by":"crossref","unstructured":"Wang, Y., Xu, Z., Shen, H., Cheng, B., & Yang, L. (2020). Centermask: single shot instance segmentation with point representation. In: proceedings of the IEEE\/CVF computer vision and pattern recognition (CVPR), pp. 9313\u20139321.","DOI":"10.1109\/CVPR42600.2020.00933"},{"key":"2261_CR67","unstructured":"Wei, Y., Hu, D., Tian, Y., & Li, X. (2022). Learning in audio-visual context: A review, analysis, and new perspective. arXiv preprint arXiv:2208.09579."},{"key":"2261_CR68","doi-asserted-by":"crossref","unstructured":"Wu, J., Jiang, Y., Sun, P., Yuan, Z., & Luo, P. (2022). Language as queries for referring video object segmentation. In: proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp. 4974\u20134984.","DOI":"10.1109\/CVPR52688.2022.00492"},{"key":"2261_CR69","doi-asserted-by":"crossref","unstructured":"Wu, Y., & Yang, Y. (2021). Exploring heterogeneous clues for weakly-supervised audio-visual video parsing. In: proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp. 1326\u20131335.","DOI":"10.1109\/CVPR46437.2021.00138"},{"key":"2261_CR70","doi-asserted-by":"crossref","unstructured":"Wu, Y., Zhu, L., Yan, Y., & Yang, Y. (2019). Dual attention matching for audio-visual event localization. In: proceedings of the IEEE international conference on computer vision (ICCV), pp. 6292\u20136300.","DOI":"10.1109\/ICCV.2019.00639"},{"key":"2261_CR71","unstructured":"Xie, E., Wang, W., Yu, Z., Anandkumar, A., Alvarez, J.M., Luo, P. (2021). Segformer: Simple and efficient design for semantic segmentation with transformers. Adv. Neural Inform. Process. Syst. (NeurIPS), pp. 12077\u201312090."},{"key":"2261_CR72","doi-asserted-by":"crossref","unstructured":"Xu, H., Zeng, R., Wu, Q., Tan, M., Gan, C. (2020). Cross-modal relation-aware networks for audio-visual event localization. In: proceedings of the 28th ACM international conference on multimedia (ACM MM), pp. 3893\u20133901.","DOI":"10.1145\/3394171.3413581"},{"key":"2261_CR73","unstructured":"Yang, Z., Wei, Y., & Yang, Y. (2021). Associating objects with transformers for video object segmentation. Adv. Neural Inform. Process. Syst. (NeurIPS), pp. 1\u201320."},{"key":"2261_CR74","doi-asserted-by":"crossref","unstructured":"Yu, J., Cheng, Y., Zhao, R.W., Feng, R., & Zhang, Y. (2022). MM-Pyramid: Multimodal pyramid attentional network for audio-visual event localization and video parsing. In: proceedings of the 30th ACM international conference on multimedia (ACM MM).","DOI":"10.1145\/3503161.3547869"},{"key":"2261_CR75","unstructured":"Zhang, J., Xie, J., Barnes, N., & Li, P. (2021). Learning generative vision transformer with energy-based latent space for saliency prediction. Advances in Neural Information Processing Systems (NeurIPS)."},{"key":"2261_CR76","doi-asserted-by":"crossref","unstructured":"Zhang, L., Zhang, J., Lin, Z., M\u011bch, R., Lu, H., & He, Y. (2020). Unsupervised video object segmentation with joint hotspot tracking. In: proceedings of the European conference on computer vision (ECCV), pp. 490\u2013506.","DOI":"10.1007\/978-3-030-58568-6_29"},{"key":"2261_CR77","doi-asserted-by":"crossref","unstructured":"Zhao, H., Gan, C., Rouditchenko, A., Vondrick, C., McDermott, J., & Torralba, A. (2018). The sound of pixels. In: proceedings of the European conference on computer vision (ECCV), pp. 570\u2013586.","DOI":"10.1007\/978-3-030-01246-5_35"},{"key":"2261_CR78","doi-asserted-by":"crossref","unstructured":"Zhao, H., Gan, C., Ma, W.C., & Torralba, A. (2019). The sound of motions. In: proceedings of the IEEE\/CVF international conference on computer vision (ICCV), pp. 1735\u20131744.","DOI":"10.1109\/ICCV.2019.00182"},{"key":"2261_CR79","doi-asserted-by":"crossref","unstructured":"Zhou, B., Khosla, A., Lapedriza, A., Oliva, A., & Torralba, A. (2016). Learning deep features for discriminative localization. In: proceedings of the IEEE\/CVF conference on computer vision and pattern recognition (CVPR), pp. 2921\u20132929.","DOI":"10.1109\/CVPR.2016.319"},{"key":"2261_CR80","doi-asserted-by":"crossref","unstructured":"Zhou, J., Zheng, L., Zhong, Y., Hao, S., Wang, M. (2021). Positive sample propagation along the audio-visual event line. In: proceedings of the IEEE\/cvf conference on computer vision and pattern recognition (CVPR), pp. 8436\u20138444.","DOI":"10.1109\/CVPR46437.2021.00833"},{"key":"2261_CR81","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2022.3223688","author":"J Zhou","year":"2022","unstructured":"Zhou, J., Guo, D., & Wang, M. (2022). Contrastive positive sample propagation along the audio-visual event line. IEEE Trans. Patt. Anal. Mach. Intell.[SPACE]https:\/\/doi.org\/10.1109\/TPAMI.2022.3223688","journal-title":"IEEE Trans. Patt. Anal. Mach. Intell."},{"key":"2261_CR82","doi-asserted-by":"crossref","unstructured":"Zhou, J., Wang, J., Zhang, J., Sun, W., Zhang, J., Birchfield, S., Guo, D., Kong, L., Wang, M., & Zhong, Y. (2022b). Audio\u2013visual segmentation. In: proceedings of the European conference on computer vision (ECCV), pp. 386\u2013403.","DOI":"10.1007\/978-3-031-19836-6_22"},{"key":"2261_CR83","unstructured":"Zhou, J., Guo, D., Zhong, Y., & Wang, M. (2023). Improving audio-visual video parsing with pseudo visual labels. arXiv preprint arXiv:2303.02344."},{"key":"2261_CR84","doi-asserted-by":"crossref","unstructured":"Zhou, J., Guo, D., Mao, Y., Zhong, Y., Chang, X., & Wang, M. (2024a). Label-anticipated event disentanglement for audio-visual video parsing. In: European conference on computer vision (ECCV), pp. 1\u201322.","DOI":"10.1007\/978-3-031-72684-2_3"},{"key":"2261_CR85","doi-asserted-by":"crossref","unstructured":"Zhou, J., Guo, D., Zhong, Y., & Wang, M. (2024b). Advancing weakly-supervised audio-visual video parsing via segment-wise pseudo labeling. Int. J. Comput. Vis. (IJCV) pp. 1\u201322.","DOI":"10.1007\/s11263-024-02142-3"}],"container-title":["International Journal of Computer Vision"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-024-02261-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11263-024-02261-x\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11263-024-02261-x.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,3,30]],"date-time":"2025-03-30T22:03:31Z","timestamp":1743372211000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11263-024-02261-x"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,15]]},"references-count":85,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2025,4]]}},"alternative-id":["2261"],"URL":"https:\/\/doi.org\/10.1007\/s11263-024-02261-x","relation":{},"ISSN":["0920-5691","1573-1405"],"issn-type":[{"value":"0920-5691","type":"print"},{"value":"1573-1405","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,15]]},"assertion":[{"value":"7 August 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 September 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"15 October 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}