{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,5,12]],"date-time":"2025-05-12T13:29:43Z","timestamp":1747056583308,"version":"3.40.3"},"publisher-location":"Cham","reference-count":30,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031533105"},{"type":"electronic","value":"9783031533112"}],"license":[{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024]]},"DOI":"10.1007\/978-3-031-53311-2_3","type":"book-chapter","created":{"date-parts":[[2024,1,27]],"date-time":"2024-01-27T21:37:36Z","timestamp":1706391456000},"page":"28-42","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Multi-task Collaborative Network for\u00a0Image-Text Retrieval"],"prefix":"10.1007","author":[{"given":"Xueyang","family":"Qin","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lishuang","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jing","family":"Hao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Meiling","family":"Ge","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiayi","family":"Huang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Guangyao","family":"Pang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,1,28]]},"reference":[{"doi-asserted-by":"crossref","unstructured":"Anderson, P., et al.: Bottom-up and top-down attention for image captioning and visual question answering. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6077\u20136086 (2018)","key":"3_CR1","DOI":"10.1109\/CVPR.2018.00636"},{"key":"3_CR2","doi-asserted-by":"publisher","first-page":"108753","DOI":"10.1016\/j.patcog.2022.108753","volume":"129","author":"J Chen","year":"2022","unstructured":"Chen, J., Yang, L., Tan, L., Xu, R.: Orthogonal channel attention-based multi-task learning for multi-view facial expression recognition. Pattern Recogn. 129, 108753 (2022)","journal-title":"Pattern Recogn."},{"doi-asserted-by":"crossref","unstructured":"Cheng, Y., Zhu, X., Qian, J., Wen, F., Liu, P.: Cross-modal graph matching network for image-text retrieval. ACM Trans. Multimedia Comput. Commun. Appl. (TOMM) 18(4), 1\u201323 (2022)","key":"3_CR3","DOI":"10.1145\/3499027"},{"issue":"4","key":"3_CR4","doi-asserted-by":"publisher","first-page":"1173","DOI":"10.1109\/TCSVT.2019.2900171","volume":"30","author":"J Chi","year":"2019","unstructured":"Chi, J., Peng, Y.: Zero-shot cross-media embedding learning with dual adversarial distribution network. IEEE Trans. Circuits Syst. Video Technol. 30(4), 1173\u20131187 (2019)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"issue":"3","key":"3_CR5","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3570640","volume":"41","author":"Y Deng","year":"2023","unstructured":"Deng, Y., Zhang, W., Xu, W., Lei, W., Chua, T.S., Lam, W.: A unified multi-task learning framework for multi-goal conversational recommender systems. ACM Trans. Inf. Syst. 41(3), 1\u201325 (2023)","journal-title":"ACM Trans. Inf. Syst."},{"doi-asserted-by":"crossref","unstructured":"Diao, H., Zhang, Y., Ma, L., Lu, H.: Similarity reasoning and filtration for image-text matching. In: Proceedings of the AAAI Conference on Artificial Intelligence, pp. 1218\u20131226 (2021)","key":"3_CR6","DOI":"10.1609\/aaai.v35i2.16209"},{"doi-asserted-by":"crossref","unstructured":"Gao, Q., Lian, H., Wang, Q., Sun, G.: Cross-modal subspace clustering via deep canonical correlation analysis. In: Proceedings of the AAAI Conference on Artificial Intelligence, pp. 3938\u20133945 (2020)","key":"3_CR7","DOI":"10.1609\/aaai.v34i04.5808"},{"doi-asserted-by":"crossref","unstructured":"Ji, Z., Chen, K., Wang, H.: Step-wise hierarchical alignment network for image-text matching. In: Proceedings of the 31th International Joint Conference on Artificial Intelligence (2021)","key":"3_CR8","DOI":"10.24963\/ijcai.2021\/106"},{"unstructured":"Kenton, J.D.M.W.C., Toutanova, L.K.: Bert: Pre-training of deep bidirectional transformers for language understanding. In: Proceedings of NAACL-HLT, pp. 4171\u20134186 (2019)","key":"3_CR9"},{"doi-asserted-by":"crossref","unstructured":"Lee, K.H., Chen, X., Hua, G., Hu, H., He, X.: Stacked cross attention for image-text matching. In: Proceedings of the European Conference on Computer Vision, pp. 201\u2013216 (2018)","key":"3_CR10","DOI":"10.1007\/978-3-030-01225-0_13"},{"doi-asserted-by":"crossref","unstructured":"Li, K., Zhang, Y., Li, K., Li, Y., Fu, Y.: Visual semantic reasoning for image-text matching. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 4654\u20134662 (2019)","key":"3_CR11","DOI":"10.1109\/ICCV.2019.00475"},{"issue":"1","key":"3_CR12","doi-asserted-by":"publisher","first-page":"102432","DOI":"10.1016\/j.ipm.2020.102432","volume":"58","author":"W Li","year":"2021","unstructured":"Li, W., Yang, S., Wang, Y., Song, D., Li, X.: Multi-level similarity learning for image-text retrieval. Inf. Process. Manage. 58(1), 102432 (2021)","journal-title":"Inf. Process. Manage."},{"doi-asserted-by":"crossref","unstructured":"Liu, C., Mao, Z., Liu, A.A., Zhang, T., Wang, B., Zhang, Y.: Focus your attention: a bidirectional focal attention network for image-text matching. In: Proceedings of the 27th ACM International Conference on Multimedia, pp. 3\u201311 (2019)","key":"3_CR13","DOI":"10.1145\/3343031.3350869"},{"issue":"2","key":"3_CR14","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3560485","volume":"41","author":"K Liu","year":"2023","unstructured":"Liu, K., Xue, F., Guo, D., Wu, L., Li, S., Hong, R.: MEGCF: multimodal entity graph collaborative filtering for personalized recommendation. ACM Trans. Inf. Syst. 41(2), 1\u201327 (2023)","journal-title":"ACM Trans. Inf. Syst."},{"doi-asserted-by":"crossref","unstructured":"Peng, Y., Qi, J.: Cm-GANs: cross-modal generative adversarial networks for common representation learning. ACM Trans. Multimedia Comput. Commun. Appl. (TOMM) 15(1), 1\u201324 (2019)","key":"3_CR15","DOI":"10.1145\/3284750"},{"key":"3_CR16","doi-asserted-by":"publisher","first-page":"105923","DOI":"10.1016\/j.engappai.2023.105923","volume":"120","author":"X Qin","year":"2023","unstructured":"Qin, X., Li, L., Hao, F., Pang, G., Wang, Z.: Cross-modal information balance-aware reasoning network for image-text retrieval. Eng. Appl. Artif. Intell. 120, 105923 (2023)","journal-title":"Eng. Appl. Artif. Intell."},{"doi-asserted-by":"publisher","unstructured":"Qin, X., Li, L., Pang, G.: Multi-scale motivated neural network for image-text matching. Multimedia Tools Appl. 1\u201325 (2023). https:\/\/doi.org\/10.1007\/s11042-023-15321-0","key":"3_CR17","DOI":"10.1007\/s11042-023-15321-0"},{"doi-asserted-by":"crossref","unstructured":"Sarafianos, N., Xu, X., Kakadiaris, I.A.: Adversarial representation learning for text-to-image matching. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 5814\u20135824 (2019)","key":"3_CR18","DOI":"10.1109\/ICCV.2019.00591"},{"doi-asserted-by":"crossref","unstructured":"Tang, H., Liu, J., Zhao, M., Gong, X.: Progressive layered extraction (PLE): a novel multi-task learning (mtl) model for personalized recommendations. In: Proceedings of the 14th ACM Conference on Recommender Systems, pp. 269\u2013278 (2020)","key":"3_CR19","DOI":"10.1145\/3383313.3412236"},{"key":"3_CR20","first-page":"1","volume":"25","author":"Z Tao","year":"2022","unstructured":"Tao, Z., Liu, X., Xia, Y., Wang, X., Yang, L., Huang, X., Chua, T.S.: Self-supervised learning for multimedia recommendation. IEEE Trans. Multimedia 25, 1\u201310 (2022)","journal-title":"IEEE Trans. Multimedia"},{"issue":"3","key":"3_CR21","doi-asserted-by":"publisher","first-page":"103280","DOI":"10.1016\/j.ipm.2023.103280","volume":"60","author":"Y Wang","year":"2023","unstructured":"Wang, Y., Su, Y., Li, W., Sun, Z., Wei, Z., Nie, J., Li, X., Liu, A.A.: Rare-aware attention network for image-text matching. Inf. Process. Manage. 60(3), 103280 (2023)","journal-title":"Inf. Process. Manage."},{"doi-asserted-by":"crossref","unstructured":"Wei, X., Zhang, T., Li, Y., Zhang, Y., Wu, F.: Multi-modality cross attention network for image and sentence matching. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10941\u201310950 (2020)","key":"3_CR22","DOI":"10.1109\/CVPR42600.2020.01095"},{"issue":"1","key":"3_CR23","doi-asserted-by":"publisher","first-page":"388","DOI":"10.1109\/TCSVT.2021.3060713","volume":"32","author":"J Wu","year":"2021","unstructured":"Wu, J., Wu, C., Lu, J., Wang, L., Cui, X.: Region reinforcement network with topic constraint for image-text matching. IEEE Trans. Circuits Syst. Video Technol. 32(1), 388\u2013397 (2021)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"issue":"12","key":"3_CR24","doi-asserted-by":"publisher","first-page":"5412","DOI":"10.1109\/TNNLS.2020.2967597","volume":"31","author":"X Xu","year":"2020","unstructured":"Xu, X., Wang, T., Yang, Y., Zuo, L., Shen, F., Shen, H.T.: Cross-modal attention with semantic consistence for image-text matching. IEEE Trans. Neural Netw. Learn. Syst. 31(12), 5412\u20135425 (2020)","journal-title":"IEEE Trans. Neural Netw. Learn. Syst."},{"key":"3_CR25","doi-asserted-by":"publisher","first-page":"2015","DOI":"10.1109\/TASLP.2022.3178204","volume":"30","author":"B Yang","year":"2022","unstructured":"Yang, B., Wu, L., Zhu, J., Shao, B., Lin, X., Liu, T.Y.: Multimodal sentiment analysis with two-phase multi-task learning. IEEE\/ACM Trans. Audio Speech Lang. Process. 30, 2015\u20132024 (2022)","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"3_CR26","doi-asserted-by":"publisher","first-page":"108401","DOI":"10.1016\/j.patcog.2021.108401","volume":"123","author":"W Yu","year":"2022","unstructured":"Yu, W., Xu, H.: Co-attentive multi-task convolutional neural network for facial expression recognition. Pattern Recogn. 123, 108401 (2022)","journal-title":"Pattern Recogn."},{"doi-asserted-by":"crossref","unstructured":"Yu, W., Xu, H., Yuan, Z., Wu, J.: Learning modality-specific representations with self-supervised multi-task learning for multimodal sentiment analysis. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 35, pp. 10790\u201310797 (2021)","key":"3_CR27","DOI":"10.1609\/aaai.v35i12.17289"},{"doi-asserted-by":"crossref","unstructured":"Yuan, H., Huang, Y., Zhang, D., Chen, Z., Cheng, W., Wang, L.: VSR++: improving visual semantic reasoning for fine-grained image-text matching. In: Proceedings of the 25th International Conference on Pattern Recognition, pp. 3728\u20133735 (2021)","key":"3_CR28","DOI":"10.1109\/ICPR48806.2021.9413223"},{"doi-asserted-by":"crossref","unstructured":"Zhang, Q., Lei, Z., Zhang, Z., Li, S.Z.: Context-aware attention network for image-text retrieval. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3536\u20133545 (2020)","key":"3_CR29","DOI":"10.1109\/CVPR42600.2020.00359"},{"key":"3_CR30","doi-asserted-by":"publisher","first-page":"110280","DOI":"10.1016\/j.knosys.2023.110280","volume":"263","author":"G Zhao","year":"2023","unstructured":"Zhao, G., Zhang, C., Shang, H., Wang, Y., Zhu, L., Qian, X.: Generative label fused network for image-text matching. Knowl.-Based Syst. 263, 110280 (2023)","journal-title":"Knowl.-Based Syst."}],"container-title":["Lecture Notes in Computer Science","MultiMedia Modeling"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-53311-2_3","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,3,12]],"date-time":"2024-03-12T15:21:45Z","timestamp":1710256905000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-53311-2_3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024]]},"ISBN":["9783031533105","9783031533112"],"references-count":30,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-53311-2_3","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2024]]},"assertion":[{"value":"28 January 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"MMM","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Multimedia Modeling","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Amsterdam","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"The Netherlands","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 January 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2 February 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"30","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"mmm2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"ConfTool Pro","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"297","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"112","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"38% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.2","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.2","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}