{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,2]],"date-time":"2026-06-02T09:14:38Z","timestamp":1780391678110,"version":"3.54.1"},"publisher-location":"Cham","reference-count":36,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031200762","type":"print"},{"value":"9783031200779","type":"electronic"}],"license":[{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2022]]},"DOI":"10.1007\/978-3-031-20077-9_36","type":"book-chapter","created":{"date-parts":[[2022,11,5]],"date-time":"2022-11-05T16:21:52Z","timestamp":1667665312000},"page":"612-628","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":37,"title":["Weakly Supervised Object Localization via\u00a0Transformer with\u00a0Implicit Spatial Calibration"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5693-1993","authenticated-orcid":false,"given":"Haotian","family":"Bai","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9511-7532","authenticated-orcid":false,"given":"Ruimao","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3676-2544","authenticated-orcid":false,"given":"Jiong","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xiang","family":"Wan","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2022,11,6]]},"reference":[{"key":"36_CR1","doi-asserted-by":"crossref","unstructured":"Bourigault, S., Lagnier, C., Lamprier, S., Denoyer, L., Gallinari, P.: Learning social network embeddings for predicting information diffusion. In: Proceedings of the 7th ACM International Conference on Web Search and Data Mining, pp. 393\u2013402 (2014)","DOI":"10.1145\/2556195.2556216"},{"key":"36_CR2","doi-asserted-by":"crossref","unstructured":"Chen, Z., et al.: On awakening the local continuity of transformer for weakly supervised object localization. In: Proceedings of the AAAI Conference on Artificial Intelligence (2022)","DOI":"10.1609\/aaai.v36i1.19918"},{"issue":"5","key":"36_CR3","doi-asserted-by":"publisher","first-page":"907","DOI":"10.1109\/JPROC.2018.2799702","volume":"106","author":"G Cheung","year":"2018","unstructured":"Cheung, G., Magli, E., Tanaka, Y., Ng, M.K.: Graph spectral image processing. Proc. IEEE 106(5), 907\u2013930 (2018)","journal-title":"Proc. IEEE"},{"key":"36_CR4","doi-asserted-by":"crossref","unstructured":"Choe, J., Oh, S.J., Lee, S., Chun, S., Akata, Z., Shim, H.: Evaluating weakly supervised object localization methods right. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 3133\u20133142 (2020)","DOI":"10.1109\/CVPR42600.2020.00320"},{"key":"36_CR5","doi-asserted-by":"crossref","unstructured":"Choe, J., Shim, H.: Attention-based dropout layer for weakly supervised object localization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 2219\u20132228 (2019)","DOI":"10.1109\/CVPR.2019.00232"},{"issue":"1","key":"36_CR6","doi-asserted-by":"publisher","first-page":"92","DOI":"10.1109\/TPAMI.2012.63","volume":"35","author":"S Gao","year":"2013","unstructured":"Gao, S., Tsang, I.W.H., Chia, L.T.: Laplacian sparse coding, hypergraph laplacian sparse coding, and applications. IEEE Trans. Pattern Anal. Mach. Intell. 35(1), 92\u2013104 (2013)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"36_CR7","doi-asserted-by":"crossref","unstructured":"Gao, W., et al.: Ts-cam: token semantic coupled attention map for weakly supervised object localization. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2886\u20132895 (2021)","DOI":"10.1109\/ICCV48922.2021.00288"},{"key":"36_CR8","doi-asserted-by":"crossref","unstructured":"Gulati, A., et al.: Conformer: convolution-augmented transformer for speech recognition. arXiv preprint arXiv:2005.08100 (2020)","DOI":"10.21437\/Interspeech.2020-3015"},{"key":"36_CR9","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Delving deep into rectifiers: surpassing human-level performance on imagenet classification. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 1026\u20131034 (2015)","DOI":"10.1109\/ICCV.2015.123"},{"key":"36_CR10","doi-asserted-by":"crossref","unstructured":"Kim, E., Kim, S., Lee, J., Kim, H., Yoon, S.: Bridging the gap between classification and localization for weakly supervised object localization. arXiv preprint arXiv:2204.00220 (2022)","DOI":"10.1109\/CVPR52688.2022.01386"},{"key":"36_CR11","unstructured":"Kondor, R.I., Lafferty, J.: Diffusion kernels on graphs and other discrete structures. In: Proceedings of the 19th International Conference on Machine Learning, vol. 2002, pp. 315\u2013322 (2002)"},{"key":"36_CR12","doi-asserted-by":"crossref","unstructured":"Liu, Z., et al.: Swin transformer: hierarchical vision transformer using shifted windows. 2021 IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 9992\u201310002 (2021)","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"36_CR13","doi-asserted-by":"crossref","unstructured":"Liu, Z., Li, X., Luo, P., Loy, C.C., Tang, X.: Semantic image segmentation via deep parsing network. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 1377\u20131385 (2015)","DOI":"10.1109\/ICCV.2015.162"},{"issue":"8","key":"36_CR14","doi-asserted-by":"publisher","first-page":"1814","DOI":"10.1109\/TPAMI.2017.2737535","volume":"40","author":"Z Liu","year":"2017","unstructured":"Liu, Z., Li, X., Luo, P., Loy, C.C., Tang, X.: Deep learning Markov random field for semantic segmentation. IEEE Trans. Pattern Anal. Mach. Intell. 40(8), 1814\u20131828 (2017)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"36_CR15","unstructured":"Loshchilov, I., Hutter, F.: Fixing weight decay regularization in adam (2018)"},{"key":"36_CR16","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"481","DOI":"10.1007\/978-3-030-58574-7_29","volume-title":"Computer Vision \u2013 ECCV 2020","author":"W Lu","year":"2020","unstructured":"Lu, W., Jia, X., Xie, W., Shen, L., Zhou, Y., Duan, J.: Geometry constrained weakly supervised object localization. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12371, pp. 481\u2013496. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58574-7_29"},{"issue":"6","key":"36_CR17","doi-asserted-by":"publisher","first-page":"1051","DOI":"10.1109\/TKDE.2011.18","volume":"24","author":"H Ma","year":"2011","unstructured":"Ma, H., King, I., Lyu, M.R.: Mining web graphs for recommendations. IEEE Trans. Knowl. Data Eng. 24(6), 1051\u20131064 (2011)","journal-title":"IEEE Trans. Knowl. Data Eng."},{"key":"36_CR18","doi-asserted-by":"crossref","unstructured":"Mai, J., Yang, M., Luo, W.: Erasing integrated learning: a simple yet effective approach for weakly supervised object localization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 8766\u20138775 (2020)","DOI":"10.1109\/CVPR42600.2020.00879"},{"key":"36_CR19","doi-asserted-by":"publisher","first-page":"1774","DOI":"10.1109\/TIP.2022.3145238","volume":"31","author":"M Meng","year":"2022","unstructured":"Meng, M., Zhang, T., Yang, W., Zhao, J., Zhang, Y., Wu, F.: Diverse complementary part mining for weakly supervised object localization. IEEE Trans. Image Process. 31, 1774\u20131788 (2022)","journal-title":"IEEE Trans. Image Process."},{"key":"36_CR20","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"504","DOI":"10.1007\/3-540-16042-6_29","volume-title":"Foundations of Software Technology and Theoretical Computer Science","author":"V Pan","year":"1985","unstructured":"Pan, V.: Fast and efficient parallel algorithms for the exact inversion of integer matrices. In: Maheshwari, S.N. (ed.) FSTTCS 1985. LNCS, vol. 206, pp. 504\u2013521. Springer, Heidelberg (1985). https:\/\/doi.org\/10.1007\/3-540-16042-6_29"},{"key":"36_CR21","doi-asserted-by":"crossref","unstructured":"Pan, V., Reif, J.: Efficient parallel solution of linear systems. In: Proceedings of the Seventeenth Annual ACM Symposium on Theory of Computing, pp. 143\u2013152 (1985)","DOI":"10.1145\/22145.22161"},{"issue":"12","key":"36_CR22","doi-asserted-by":"publisher","first-page":"1991","DOI":"10.1101\/gr.077693.108","volume":"18","author":"Y Qi","year":"2008","unstructured":"Qi, Y., Suhail, Y., Lin, Y.Y., Boeke, J.D., Bader, J.S.: Finding friends and enemies in an enemies-only network: a graph diffusion kernel for predicting novel genetic interactions and co-complex membership from yeast genetic interactions. Genome Res. 18(12), 1991\u20132004 (2008)","journal-title":"Genome Res."},{"issue":"3","key":"36_CR23","doi-asserted-by":"publisher","first-page":"211","DOI":"10.1007\/s11263-015-0816-y","volume":"115","author":"O Russakovsky","year":"2015","unstructured":"Russakovsky, O., et al.: Imagenet large scale visual recognition challenge. Int. J. Comput. Vision 115(3), 211\u2013252 (2015)","journal-title":"Int. J. Comput. Vision"},{"key":"36_CR24","unstructured":"Sharir, G., Noy, A., Zelnik-Manor, L.: An image is worth 16$$\\times $$16 words, what is a video worth? arXiv preprint arXiv:2103.13915 (2021)"},{"key":"36_CR25","doi-asserted-by":"crossref","unstructured":"Singh, K.K., Lee, Y.J.: Hide-and-seek: forcing a network to be meticulous for weakly-supervised object and action localization (2017)","DOI":"10.1109\/ICCV.2017.381"},{"key":"36_CR26","unstructured":"Touvron, H., Cord, M., Douze, M., Massa, F., Sablayrolles, A., J\u00e9gou, H.: Training data-efficient image transformers & distillation through attention. In: International Conference on Machine Learning, pp. 10347\u201310357. PMLR (2021)"},{"key":"36_CR27","doi-asserted-by":"crossref","unstructured":"Wei, J., Wang, Q., Li, Z., Wang, S., Zhou, S.K., Cui, S.: Shallow feature matters for weakly supervised object localization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 5993\u20136001 (2021)","DOI":"10.1109\/CVPR46437.2021.00593"},{"key":"36_CR28","doi-asserted-by":"crossref","unstructured":"Wei, J., Wang, S., Zhou, S.K., Cui, S., Li, Z.: Weakly supervised object localization through inter-class feature similarity and intra-class appearance consistency. In: European Conference on Computer Vision. Springer, Heidelberg (2022)","DOI":"10.1007\/978-3-031-20056-4_12"},{"key":"36_CR29","unstructured":"Welinder, P., et al.: Caltech-ucsd birds 200. Technical report (2010)"},{"key":"36_CR30","doi-asserted-by":"crossref","unstructured":"Xue, H., Liu, C., Wan, F., Jiao, J., Ji, X., Ye, Q.: Danet: divergent activation for weakly supervised object localization. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6589\u20136598 (2019)","DOI":"10.1109\/ICCV.2019.00669"},{"key":"36_CR31","doi-asserted-by":"crossref","unstructured":"Yun, S., Han, D., Oh, S.J., Chun, S., Choe, J., Yoo, Y.: Cutmix: regularization strategy to train strong classifiers with localizable features. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6023\u20136032 (2019)","DOI":"10.1109\/ICCV.2019.00612"},{"key":"36_CR32","doi-asserted-by":"crossref","unstructured":"Zhang, C.L., Cao, Y.H., Wu, J.: Rethinking the route towards weakly supervised object localization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13460\u201313469 (2020)","DOI":"10.1109\/CVPR42600.2020.01347"},{"key":"36_CR33","doi-asserted-by":"crossref","unstructured":"Zhang, X., Wei, Y., Feng, J., Yang, Y., Huang, T.S.: Adversarial complementary learning for weakly supervised object localization. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1325\u20131334 (2018)","DOI":"10.1109\/CVPR.2018.00144"},{"key":"36_CR34","doi-asserted-by":"crossref","unstructured":"Zhang, X., Wei, Y., Kang, G., Yang, Y., Huang, T.: Self-produced guidance for weakly-supervised object localization. In: Proceedings of the European Conference on Computer Vision (ECCV), pp. 597\u2013613 (2018)","DOI":"10.1007\/978-3-030-01258-8_37"},{"key":"36_CR35","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"271","DOI":"10.1007\/978-3-030-58529-7_17","volume-title":"Computer Vision \u2013 ECCV 2020","author":"X Zhang","year":"2020","unstructured":"Zhang, X., Wei, Y., Yang, Y.: Inter-image communication for weakly supervised localization. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12364, pp. 271\u2013287. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58529-7_17"},{"key":"36_CR36","doi-asserted-by":"crossref","unstructured":"Zhou, B., Khosla, A., Lapedriza, A., Oliva, A., Torralba, A.: Learning deep features for discriminative localization. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 2921\u20132929 (2016)","DOI":"10.1109\/CVPR.2016.319"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2022"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-20077-9_36","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,11,8]],"date-time":"2022-11-08T00:18:51Z","timestamp":1667866731000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-20077-9_36"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022]]},"ISBN":["9783031200762","9783031200779"],"references-count":36,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-20077-9_36","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022]]},"assertion":[{"value":"6 November 2022","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Tel Aviv","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Israel","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2022","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"23 October 2022","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27 October 2022","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2022","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2022.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Double-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"CMT","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"5804","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"1645","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"28% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.21","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3.91","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}