{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,9]],"date-time":"2026-06-09T15:50:57Z","timestamp":1781020257590,"version":"3.54.1"},"publisher-location":"Cham","reference-count":29,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031790409","type":"print"},{"value":"9783031790416","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-79041-6_21","type":"book-chapter","created":{"date-parts":[[2025,1,31]],"date-time":"2025-01-31T11:47:58Z","timestamp":1738324078000},"page":"261-274","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["A Contrastive Meta-learning Approach with\u00a0Isotropic Sparse Decomposition for\u00a0Scalable Audio-Visual Learning"],"prefix":"10.1007","author":[{"given":"Dhruv","family":"Dixit","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Paritosh","family":"Pandey","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Raman","family":"Jha","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Pranshul","family":"Bhatnagar","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1665-8101","authenticated-orcid":false,"given":"Shashank Mouli","family":"Satapathy","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,2,1]]},"reference":[{"key":"21_CR1","unstructured":"Abbasi, A., Monadjemi, A., Fang, L., Rabbani, H., Noormohammadi, N., Zhang, Y.: Multiscale sparsifying transform learning for image denoising. arXiv preprint arXiv:2003.11265 (2020)"},{"key":"21_CR2","doi-asserted-by":"crossref","unstructured":"Ahmad, M., Ghous, U., Usama, M., Mazzara, M.: Waveformer: spectral\u2013spatial wavelet transformer for hyperspectral image classification. IEEE Geosci. Remote Sens. Lett. (2024)","DOI":"10.1109\/LGRS.2024.3353909"},{"key":"21_CR3","doi-asserted-by":"crossref","unstructured":"Chen, J., Zhang, R., Mao, Y., Xu, J.: Contrastnet: a contrastive learning framework for few-shot text classification. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a036, pp. 10492\u201310500 (2022)","DOI":"10.1609\/aaai.v36i10.21292"},{"key":"21_CR4","first-page":"16664","volume":"35","author":"S Chen","year":"2022","unstructured":"Chen, S., et al.: Adaptformer: adapting vision transformers for scalable visual recognition. Adv. Neural. Inf. Process. Syst. 35, 16664\u201316678 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"21_CR5","doi-asserted-by":"crossref","unstructured":"Dong, B., Wang, Y., Sun, H., Wang, Y., Hashemi, A., Du, Z.: CML: a contrastive meta learning method to estimate human label confidence scores and reduce data collection cost. In: Proceedings of the Fifth Workshop on e-Commerce and NLP (ECNLP 5), pp. 35\u201343 (2022)","DOI":"10.18653\/v1\/2022.ecnlp-1.5"},{"issue":"3","key":"21_CR6","doi-asserted-by":"publisher","first-page":"613","DOI":"10.1109\/18.382009","volume":"41","author":"DL Donoho","year":"1995","unstructured":"Donoho, D.L.: De-noising by soft-thresholding. IEEE Trans. Inf. Theory 41(3), 613\u2013627 (1995)","journal-title":"IEEE Trans. Inf. Theory"},{"key":"21_CR7","doi-asserted-by":"crossref","unstructured":"Gao, R., Oh, T.H., Grauman, K., Torresani, L.: Listen to look: action recognition by previewing audio. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10457\u201310467 (2020)","DOI":"10.1109\/CVPR42600.2020.01047"},{"key":"21_CR8","unstructured":"Gao, Y., Fei, N., Liu, G., Lu, Z., Xiang, T.: Contrastive prototype learning with augmented embeddings for few-shot learning. In: Uncertainty in Artificial Intelligence, pp. 140\u2013150. PMLR (2021)"},{"key":"21_CR9","doi-asserted-by":"crossref","unstructured":"Geng, R., Li, B., Li, Y., Zhu, X., Jian, P., Sun, J.: Induction networks for few-shot text classification. arXiv preprint arXiv:1902.10482 (2019)","DOI":"10.18653\/v1\/D19-1403"},{"key":"21_CR10","doi-asserted-by":"crossref","unstructured":"Goyal, Y., Khot, T., Summers-Stay, D., Batra, D., Parikh, D.: Making the V in VQA matter: elevating the role of image understanding in visual question answering. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6904\u20136913 (2017)","DOI":"10.1109\/CVPR.2017.670"},{"key":"21_CR11","doi-asserted-by":"crossref","unstructured":"Han, C., Fan, Z., Zhang, D., Qiu, M., Gao, M., Zhou, A.: Meta-learning adversarial domain adaptation network for few-shot text classification. arXiv preprint arXiv:2107.12262 (2021)","DOI":"10.18653\/v1\/2021.findings-acl.145"},{"key":"21_CR12","doi-asserted-by":"crossref","unstructured":"Jin, X., et al.: MV-adapter: multimodal video transfer learning for video text retrieval. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 27144\u201327153 (2024)","DOI":"10.1109\/CVPR52733.2024.02563"},{"key":"21_CR13","doi-asserted-by":"crossref","unstructured":"Johnson, J., Hariharan, B., Van Der\u00a0Maaten, L., Fei-Fei, L., Lawrence\u00a0Zitnick, C., Girshick, R.: Clevr: a diagnostic dataset for compositional language and elementary visual reasoning. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2901\u20132910 (2017)","DOI":"10.1109\/CVPR.2017.215"},{"key":"21_CR14","doi-asserted-by":"crossref","unstructured":"Kim, J., Ma, M., Pham, T., Kim, K., Yoo, C.D.: Modality shifting attention network for multi-modal video question answering. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10106\u201310115 (2020)","DOI":"10.1109\/CVPR42600.2020.01012"},{"key":"21_CR15","doi-asserted-by":"crossref","unstructured":"Li, G., Wei, Y., Tian, Y., Xu, C., Wen, J.R., Hu, D.: Learning to answer questions in dynamic audio-visual scenarios. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 19108\u201319118 (2022)","DOI":"10.1109\/CVPR52688.2022.01852"},{"key":"21_CR16","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2024.106378","volume":"177","author":"J Li","year":"2024","unstructured":"Li, J., Cheng, B., Chen, Y., Gao, G., Shi, J., Zeng, T.: EWT: efficient wavelet-transformer for single image denoising. Neural Netw. 177, 106378 (2024)","journal-title":"Neural Netw."},{"key":"21_CR17","doi-asserted-by":"crossref","unstructured":"Li, X., et al.: Beyond RNNs: positional self-attention with co-attention for video question answering. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a033, pp. 8658\u20138665 (2019)","DOI":"10.1609\/aaai.v33i01.33018658"},{"key":"21_CR18","doi-asserted-by":"crossref","unstructured":"Lin, Y.B., Sung, Y.L., Lei, J., Bansal, M., Bertasius, G.: Vision transformers are parameter-efficient audio-visual learners. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2299\u20132309 (2023)","DOI":"10.1109\/CVPR52729.2023.00228"},{"key":"21_CR19","doi-asserted-by":"crossref","unstructured":"Liu, C., Fu, Y., Xu, C., Yang, S., Li, J., Wang, C., Zhang, L.: Learning a few-shot embedding model with contrastive learning. In: Proceedings of the AAAI conference on artificial intelligence. vol.\u00a035, pp. 8635\u20138643 (2021)","DOI":"10.1609\/aaai.v35i10.17047"},{"key":"21_CR20","doi-asserted-by":"crossref","unstructured":"Liu, X., et al.: Visual sound localization in the wild by cross-modal interference erasing. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a036, pp. 1801\u20131809 (2022)","DOI":"10.1609\/aaai.v36i2.20073"},{"issue":"07","key":"21_CR21","doi-asserted-by":"publisher","first-page":"710","DOI":"10.1109\/34.142909","volume":"14","author":"S Mallat","year":"1992","unstructured":"Mallat, S., Zhong, S.: Characterization of signals from multiscale edges. IEEE Trans. Pattern Anal. Mach. Intell. 14(07), 710\u2013732 (1992)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"21_CR22","doi-asserted-by":"crossref","unstructured":"Schwartz, I., Schwing, A.G., Hazan, T.: A simple baseline for audio-visual scene-aware dialog. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12548\u201312558 (2019)","DOI":"10.1109\/CVPR.2019.01283"},{"key":"21_CR23","unstructured":"Snell, J., Swersky, K., Zemel, R.: Prototypical networks for few-shot learning. In: Advances in Neural Information Processing Systems, vol. 30 (2017)"},{"key":"21_CR24","doi-asserted-by":"crossref","unstructured":"Sung, Y.L., Cho, J., Bansal, M.: VL-adapter: parameter-efficient transfer learning for vision-and-language tasks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5227\u20135237 (2022)","DOI":"10.1109\/CVPR52688.2022.00516"},{"key":"21_CR25","doi-asserted-by":"crossref","unstructured":"Wu, Y., Yang, Y.: Exploring heterogeneous clues for weakly-supervised audio-visual video parsing. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1326\u20131335 (2021)","DOI":"10.1109\/CVPR46437.2021.00138"},{"key":"21_CR26","doi-asserted-by":"crossref","unstructured":"Yun, H., Yu, Y., Yang, W., Lee, K., Kim, G.: Pano-avqa: Grounded audio-visual question answering on 360deg videos. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 2031\u20132041 (2021)","DOI":"10.1109\/ICCV48922.2021.00204"},{"key":"21_CR27","unstructured":"Zhang, T., Dai, D., Tuytelaars, T., Moens, M.F., Van\u00a0Gool, L.: Speech-based visual question answering. arXiv preprint arXiv:1705.00464 (2017)"},{"key":"21_CR28","doi-asserted-by":"crossref","unstructured":"Zhou, D., et al.: Sepfusion: finding optimal fusion structures for visual sound separation. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a036, pp. 3544\u20133552 (2022)","DOI":"10.1609\/aaai.v36i3.20266"},{"key":"21_CR29","doi-asserted-by":"crossref","unstructured":"Zhou, J., Zheng, L., Zhong, Y., Hao, S., Wang, M.: Positive sample propagation along the audio-visual event line. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8436\u20138444 (2021)","DOI":"10.1109\/CVPR46437.2021.00833"}],"container-title":["Communications in Computer and Information Science","Computing, Communication and Learning"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-79041-6_21","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,1,31]],"date-time":"2025-01-31T11:48:15Z","timestamp":1738324095000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-79041-6_21"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9783031790409","9783031790416"],"references-count":29,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-79041-6_21","relation":{},"ISSN":["1865-0929","1865-0937"],"issn-type":[{"value":"1865-0929","type":"print"},{"value":"1865-0937","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"1 February 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"CoCoLe","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Computing, Communication and Learning","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Warangal","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"India","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"13 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"15 September 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"3","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"cocole2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/ic-cocole.in\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}