{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T16:11:52Z","timestamp":1783786312380,"version":"3.55.0"},"publisher-location":"Singapore","reference-count":39,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819228553","type":"print"},{"value":"9789819228560","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,7,12]],"date-time":"2026-07-12T00:00:00Z","timestamp":1783814400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,7,12]],"date-time":"2026-07-12T00:00:00Z","timestamp":1783814400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-981-92-2856-0_3","type":"book-chapter","created":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T15:16:33Z","timestamp":1783782993000},"page":"32-48","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Image-Text Matching via\u00a0Graph Counterfactual Learning and\u00a0Trustworthy Alignment"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5339-0511","authenticated-orcid":false,"given":"Bo","family":"Li","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jingyi","family":"Yan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xing","family":"Wei","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Qinghui","family":"Hu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5313-6134","authenticated-orcid":false,"given":"Zhixin","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,7,12]]},"reference":[{"key":"3_CR1","doi-asserted-by":"crossref","unstructured":"Abrate, C., Bonchi, F.: Counterfactual graphs for explainable classification of brain networks. In: Proceedings of the ACM SIGKDD Conference on Knowledge Discovery & Data Mining, pp. 2495\u20132504 (2021)","DOI":"10.1145\/3447548.3467154"},{"key":"3_CR2","doi-asserted-by":"crossref","unstructured":"Anderson, P., et al.: Bottom-up and top-down attention for image captioning and visual question answering. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 6077\u20136086 (2018)","DOI":"10.1109\/CVPR.2018.00636"},{"key":"3_CR3","doi-asserted-by":"crossref","unstructured":"Chen, J., Hu, H., Wu, H., Jiang, Y., Wang, C.: Learning the best pooling strategy for visual semantic embedding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15789\u201315798 (2021)","DOI":"10.1109\/CVPR46437.2021.01553"},{"issue":"4","key":"3_CR4","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3499027","volume":"18","author":"Y Cheng","year":"2022","unstructured":"Cheng, Y., Zhu, X., Qian, J., Wen, F., Liu, P.: Cross-modal graph matching network for image-text retrieval. ACM Trans. Multimed. Comput. Commun. Appl. 18(4), 1\u201323 (2022)","journal-title":"ACM Trans. Multimed. Comput. Commun. Appl."},{"key":"3_CR5","doi-asserted-by":"crossref","unstructured":"Devlin, J., Chang, M.W., Lee, K., Toutanova, K.: Bert: pre-training of deep bidirectional transformers for language understanding. In: Proceedings of the Conference of the North American Chapter of the Association for Computational Linguistics, pp. 4171\u20134186 (2018)","DOI":"10.18653\/v1\/N19-1423"},{"key":"3_CR6","doi-asserted-by":"publisher","first-page":"2322","DOI":"10.1109\/TIP.2023.3266887","volume":"32","author":"H Diao","year":"2023","unstructured":"Diao, H., Zhang, Y., Liu, W., Ruan, X., Lu, H.: Plug-and-play regulators for image-text matching. IEEE Trans. Image Process. 32, 2322\u20132334 (2023)","journal-title":"IEEE Trans. Image Process."},{"key":"3_CR7","doi-asserted-by":"crossref","unstructured":"Diao, H., Zhang, Y., Ma, L., Lu, H.: Similarity reasoning and filtration for image-text matching. In: Proceedings of the AAAI conference on artificial intelligence, vol. 35, pp. 1218\u20131226 (2021)","DOI":"10.1609\/aaai.v35i2.16209"},{"key":"3_CR8","unstructured":"Faghri, F., Fleet, D.J., Kiros, J.R., Fidler, S.V.: VSE++: improving visual-semantic embeddings with hard negatives. In: Proceedings of the British Machine Vision Conference, pp. 1\u201313 (2018)"},{"key":"3_CR9","doi-asserted-by":"crossref","unstructured":"Fu, Z., Mao, Z., Song, Y., Zhang, Y.: Learning semantic relationship among instances for image-text matching. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15159\u201315168 (2023)","DOI":"10.1109\/CVPR52729.2023.01455"},{"issue":"2","key":"3_CR10","doi-asserted-by":"publisher","first-page":"2551","DOI":"10.1109\/TPAMI.2022.3171983","volume":"45","author":"Z Han","year":"2022","unstructured":"Han, Z., Zhang, C., Fu, H., Zhou, J.T.: Trusted multi-view classification with dynamic evidential fusion. IEEE Trans. Pattern Anal. Mach. Intell. 45(2), 2551\u20132566 (2022)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"3_CR11","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 770\u2013778 (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"3_CR12","unstructured":"Kipf, T.: Semi-supervised classification with graph convolutional networks (2016). arXiv preprint arXiv:1609.02907"},{"key":"3_CR13","doi-asserted-by":"publisher","first-page":"32","DOI":"10.1007\/s11263-016-0981-7","volume":"123","author":"R Krishna","year":"2017","unstructured":"Krishna, R., et al.: Visual genome: connecting language and vision using crowdsourced dense image annotations. Int. J. Comput. Vision 123, 32\u201373 (2017)","journal-title":"Int. J. Comput. Vision"},{"key":"3_CR14","doi-asserted-by":"crossref","unstructured":"Lee, K.H., Chen, X., Hua, G., Hu, H., He, X.: Stacked cross attention for image-text matching. In: Proceedings of the European Conference on Computer Vision, pp. 201\u2013216 (2018)","DOI":"10.1007\/978-3-030-01225-0_13"},{"key":"3_CR15","doi-asserted-by":"publisher","first-page":"276","DOI":"10.1016\/j.neunet.2023.12.018","volume":"171","author":"B Li","year":"2024","unstructured":"Li, B., Li, Z.: Large-scale cross-modal hashing with unified learning and multi-object regional correlation reasoning. Neural Netw. 171, 276\u2013292 (2024)","journal-title":"Neural Netw."},{"key":"3_CR16","doi-asserted-by":"publisher","first-page":"2497","DOI":"10.1109\/TMM.2026.3651028","volume":"28","author":"B Li","year":"2026","unstructured":"Li, B., Li, Z.: Causality-inspired graph neural networks for cross-modal retrieval. IEEE Trans. Multimedia 28, 2497\u20132509 (2026)","journal-title":"IEEE Trans. Multimedia"},{"key":"3_CR17","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2026.108902","volume":"201","author":"B Li","year":"2026","unstructured":"Li, B., Li, Z., Jiang, S., Zhang, C., Ma, H.: Twin contrastive interventional-cause hashing for unsupervised cross-modal retrieval. Neural Netw. 201, 108902 (2026)","journal-title":"Neural Netw."},{"key":"3_CR18","doi-asserted-by":"crossref","unstructured":"Li, B., Wu, Y., Li, Z.: Team HUGE: image-text matching via hierarchical and unified graph enhancing. In: Proceedings of the International Conference on Multimedia Retrieval, pp. 704\u2013712 (2024)","DOI":"10.1145\/3652583.3658001"},{"key":"3_CR19","doi-asserted-by":"publisher","DOI":"10.1016\/j.ijar.2025.109383","volume":"180","author":"B Li","year":"2025","unstructured":"Li, B., Wu, Y., Li, Z.: Efficient parameter-free adaptive hashing for large-scale cross-modal retrieval. Int. J. Approximate Reasoning 180, 109383 (2025)","journal-title":"Int. J. Approximate Reasoning"},{"key":"3_CR20","doi-asserted-by":"publisher","first-page":"34","DOI":"10.1007\/s13042-025-02818-3","volume":"17","author":"B Li","year":"2026","unstructured":"Li, B., Wu, Y., Li, Z., Wei, X.: Neighborhood-interference independent graph mechanisms for image-text matching. Int. J. Mach. Learn. Cybern. 17, 34 (2026)","journal-title":"Int. J. Mach. Learn. Cybern."},{"key":"3_CR21","doi-asserted-by":"publisher","DOI":"10.1016\/j.displa.2023.102489","volume":"79","author":"B Li","year":"2023","unstructured":"Li, B., Yao, D., Li, Z.: RICH: a rapid method for image-text cross-modal hash retrieval. Displays 79, 102489 (2023)","journal-title":"Displays"},{"issue":"2","key":"3_CR22","doi-asserted-by":"publisher","first-page":"1921","DOI":"10.1109\/TCSVT.2024.3480949","volume":"35","author":"Z Li","year":"2025","unstructured":"Li, Z., Guo, C., Wang, X., Feng, Z., Du, Z.: Selectively hard negative mining for alleviating gradient vanishing in image-text matching. IEEE Trans. Circuits Syst. Video Technol. 35(2), 1921\u20131935 (2025)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"3_CR23","doi-asserted-by":"crossref","unstructured":"Lin, T.Y., et al.: Microsoft coco: common objects in context. In: Proceedings of the European Conference on Computer Vision, pp. 740\u2013755 (2014)","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"3_CR24","doi-asserted-by":"crossref","unstructured":"Liu, C., Mao, Z., Zhang, T., Xie, H., Wang, B., Zhang, Y.: Graph structured network for image-text matching. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10921\u201310930 (2020)","DOI":"10.1109\/CVPR42600.2020.01093"},{"key":"3_CR25","unstructured":"Pan, R., Dong, J., Yang, H.: Discovering clone negatives via adaptive contrastive learning for image-text matching. In: Proceedings of the International Conference on Learning Representations, pp. 1\u201327 (2025)"},{"key":"3_CR26","doi-asserted-by":"crossref","unstructured":"Pan, Z., Wu, F., Zhang, B.: Fine-grained image-text matching by cross-modal hard aligning network. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 19275\u201319284 (2023)","DOI":"10.1109\/CVPR52729.2023.01847"},{"key":"3_CR27","doi-asserted-by":"publisher","first-page":"4515","DOI":"10.1109\/TIP.2025.3587575","volume":"34","author":"Y Qin","year":"2025","unstructured":"Qin, Y., et al.: Trustworthy visual-textual retrieval. IEEE Trans. Image Process. 34, 4515\u20134526 (2025)","journal-title":"IEEE Trans. Image Process."},{"key":"3_CR28","unstructured":"Ren, S., He, K., Girshick, R., Sun, J.: Faster R-CNN: towards real-time object detection with region proposal networks. In: Proceedings of the Advances in Neural Information Processing Systems, pp. 91\u201399 (2015)"},{"issue":"11","key":"3_CR29","doi-asserted-by":"publisher","first-page":"2673","DOI":"10.1109\/78.650093","volume":"45","author":"M Schuster","year":"1997","unstructured":"Schuster, M., Paliwal, K.: Bidirectional recurrent neural networks. IEEE Trans. Signal Process. 45(11), 2673\u20132681 (1997)","journal-title":"IEEE Trans. Signal Process."},{"key":"3_CR30","unstructured":"Sensoy, M., Kaplan, L., Kandemir, M.: Evidential deep learning to quantify classification uncertainty. In: Proceedings of the Advances in Neural Information Processing Systems, vol. 31, pp. 1\u201311 (2018)"},{"key":"3_CR31","doi-asserted-by":"crossref","unstructured":"Wang, T., Xu, X., Yang, Y., Hanjalic, A., Shen, H.T., Song, J.: Matching images and text with multi-modal tensor fusion and re-ranking. In: Proceedings of the ACM International Conference on Multimedia, pp. 12\u201320 (2019)","DOI":"10.1145\/3343031.3350875"},{"key":"3_CR32","doi-asserted-by":"crossref","unstructured":"Yao, D., Li, B., Li, Z.: Bi-directional similarity enhancement and adjustment hashing for unsupervised cross-modal retrieval. Neurocomputing, 132767 (2026)","DOI":"10.1016\/j.neucom.2026.132767"},{"key":"3_CR33","doi-asserted-by":"publisher","first-page":"67","DOI":"10.1162\/tacl_a_00166","volume":"2","author":"P Young","year":"2014","unstructured":"Young, P., Lai, A., Hodosh, M., Hockenmaier, J.: From image descriptions to visual denotations: new similarity metrics for semantic inference over event descriptions. Trans. Assoc. Comput. Linguist. 2, 67\u201378 (2014)","journal-title":"Trans. Assoc. Comput. Linguist."},{"key":"3_CR34","doi-asserted-by":"crossref","unstructured":"Zhang, H., Mao, Z., Zhang, K., Zhang, Y.: Show your faith: cross-modal confidence-aware network for image-text matching. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a036, pp. 3262\u20133270 (2022)","DOI":"10.1609\/aaai.v36i3.20235"},{"key":"3_CR35","doi-asserted-by":"crossref","unstructured":"Zhang, H., Zhang, L., Zhang, K., Mao, Z.: Identification of necessary semantic undertakers in the causal view for image-text matching. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a038, pp. 7105\u20137114 (2024)","DOI":"10.1609\/aaai.v38i7.28538"},{"key":"3_CR36","doi-asserted-by":"crossref","unstructured":"Zhou, Z., et al.: Achieving ensemble-like performance in a single model: a feature diversification framework for image-text matching. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 39, pp. 10879\u201310886 (2025)","DOI":"10.1609\/aaai.v39i10.33182"},{"key":"3_CR37","doi-asserted-by":"crossref","unstructured":"Zhou, Z., Zhang, W., Du, X., Zheng, Y., Jin, C.: Expanding the scope of negatives: boosting image-text matching with negatives distribution guided learning. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 39, pp. 10887\u201310895 (2025)","DOI":"10.1609\/aaai.v39i10.33183"},{"issue":"10","key":"3_CR38","doi-asserted-by":"publisher","first-page":"6131","DOI":"10.1109\/TCSVT.2023.3253548","volume":"33","author":"H Zhu","year":"2023","unstructured":"Zhu, H., Zhang, C., Wei, Y., Huang, S., Zhao, Y.: ESA: external space attention aggregation for image-text retrieval. IEEE Trans. Circuits Syst. Video Technol. 33(10), 6131\u20136143 (2023)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"3_CR39","doi-asserted-by":"crossref","unstructured":"Zhuang, X., Zhou, F., Li, Z.: Multi-modal sarcasm detection via knowledge-aware focused graph convolutional networks. ACM Trans. Multimedia Comput. Commun. Appl. 21(5), 143:1\u2013143:22 (2025)","DOI":"10.1145\/3722115"}],"container-title":["Lecture Notes in Computer Science","Knowledge Science, Engineering and Management"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-92-2856-0_3","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T15:16:36Z","timestamp":1783782996000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-92-2856-0_3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,7,12]]},"ISBN":["9789819228553","9789819228560"],"references-count":39,"URL":"https:\/\/doi.org\/10.1007\/978-981-92-2856-0_3","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,7,12]]},"assertion":[{"value":"12 July 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"KSEM","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Knowledge Science, Engineering and Management","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Beijing","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"China","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17 July 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"19 July 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"19","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ksem2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/ksem2026.rosc.org.cn\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}