{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,15]],"date-time":"2025-11-15T10:32:52Z","timestamp":1763202772846},"reference-count":69,"publisher":"Institute of Electronics, Information and Communications Engineers (IEICE)","issue":"1","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEICE Trans. Fundamentals"],"published-print":{"date-parts":[[2024,1,1]]},"DOI":"10.1587\/transfun.2022eap1157","type":"journal-article","created":{"date-parts":[[2023,7,5]],"date-time":"2023-07-05T22:12:06Z","timestamp":1688595126000},"page":"141-156","source":"Crossref","is-referenced-by-count":1,"title":["CCTSS: The Combination of CNN and Transformer with Shared Sublayer for Detection and Classification"],"prefix":"10.1587","volume":"E107.A","author":[{"given":"Aorui","family":"GOU","sequence":"first","affiliation":[{"name":"Academy for Engineering and Technology and the State Key Laboratory of ASIC and System, Fudan University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jingjing","family":"LIU","sequence":"additional","affiliation":[{"name":"Academy for Engineering and Technology and the State Key Laboratory of ASIC and System, Fudan University"},{"name":"Shanghai Key Laboratory of Automotive Intelligent Network Interaction Chip and System, School of Microelectronics, Shanghai University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaoxiang","family":"CHEN","sequence":"additional","affiliation":[{"name":"Academy for Engineering and Technology and the State Key Laboratory of ASIC and System, Fudan University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaoyang","family":"ZENG","sequence":"additional","affiliation":[{"name":"Academy for Engineering and Technology and the State Key Laboratory of ASIC and System, Fudan University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yibo","family":"FAN","sequence":"additional","affiliation":[{"name":"Academy for Engineering and Technology and the State Key Laboratory of ASIC and System, Fudan University"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"532","reference":[{"key":"1","doi-asserted-by":"crossref","unstructured":"[1] N. Carion, F. Massa, G. Synnaeve, N. Usunier, A. Kirillov, and S. Zagoruyko, \u201cEnd-to-end object detection with transformers,\u201d European Conference on Computer Vision, pp.213-229, Springer, 2020. 10.1007\/978-3-030-58452-8_13","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"2","doi-asserted-by":"publisher","unstructured":"[2] D. Chang, Y. Ding, J. Xie, A.K. Bhunia, X. Li, Z. Ma, M. Wu, J. Guo, and Y.-Z. Song, \u201cThe devil is in the channels: Mutual-channel loss for fine-grained image classification,\u201d IEEE Trans. Image Process., vol.29, pp.4683-4695, 2020. 10.1109\/tip.2020.2973812","DOI":"10.1109\/TIP.2020.2973812"},{"key":"3","doi-asserted-by":"crossref","unstructured":"[3] Y. Chen, Y. Bai, W. Zhang, and T. Mei, \u201cDestruction and construction learning for fine-grained image recognition,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.5157-5166, 2019. 10.1109\/cvpr.2019.00530","DOI":"10.1109\/CVPR.2019.00530"},{"key":"4","doi-asserted-by":"crossref","unstructured":"[4] Y. Chen, X. Dai, D. Chen, M. Liu, X. Dong, L. Yuan, and Z. Liu, \u201cMobile-former: Bridging mobilenet and transformer,\u201d arXiv preprint arXiv:2108.05895, 2021. 10.48550\/arXiv.2108.05895","DOI":"10.1109\/CVPR52688.2022.00520"},{"key":"5","doi-asserted-by":"crossref","unstructured":"[5] F. Chollet, \u201cXception: Deep learning with depthwise separable convolutions,\u201d Proc. IEEE Conference on Computer Vision and Pattern Recognition, pp.1251-1258, 2017. 10.1109\/cvpr.2017.195","DOI":"10.1109\/CVPR.2017.195"},{"key":"6","unstructured":"[6] M. Cuturi, \u201cSinkhorn distances: Lightspeed computation of optimal transport,\u201d Advances in Neural Information Processing Systems, vol.26, pp.2292-2300, 2013."},{"key":"7","doi-asserted-by":"crossref","unstructured":"[7] Z. Dai, B. Cai, Y. Lin, and J. Chen, \u201cUP-DETR: Unsupervised pre-training for object detection with transformers,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.1601-1610, 2021. 10.1109\/cvpr46437.2021.00165","DOI":"10.1109\/CVPR46437.2021.00165"},{"key":"8","doi-asserted-by":"publisher","unstructured":"[8] Y. Ding, Z. Ma, S. Wen, J. Xie, D. Chang, Z. Si, M. Wu, and H. Ling, \u201cAP-CNN: Weakly supervised attention pyramid convolutional neural network for fine-grained visual classification,\u201d IEEE Trans. Image Process., vol.30, pp.2826-2836, 2021. 10.1109\/tip.2021.3055617","DOI":"10.1109\/TIP.2021.3055617"},{"key":"9","doi-asserted-by":"crossref","unstructured":"[9] P. Doll\u00e1r, C. Wojek, B. Schiele, and P. Perona, \u201cPedestrian detection: A benchmark,\u201d 2009 IEEE Conference on Computer Vision and Pattern Recognition, pp.304-311, IEEE, 2009. 10.1109\/cvpr.2009.5206631","DOI":"10.1109\/CVPR.2009.5206631"},{"key":"10","unstructured":"[10] A. Dosovitskiy, L. Beyer, A. Kolesnikov, D. Weissenborn, X. Zhai, T. Unterthiner, M. Dehghani, M. Minderer, G. Heigold, S. Gelly, J. Uszkoreit, and N. Houlsby, \u201cAn image is worth 16x16 words: Transformers for image recognition at scale,\u201d arXiv preprint arXiv:2010.11929, 2020. 10.48550\/arXiv.2010.11929"},{"key":"11","doi-asserted-by":"crossref","unstructured":"[11] R. Du, D. Chang, A.K. Bhunia, J. Xie, Z. Ma, Y.-Z. Song, and J. Guo, \u201cFine-grained visual classification via progressive multi-granularity training of jigsaw patches,\u201d European Conference on Computer Vision, pp.153-168, Springer, 2020. 10.1007\/978-3-030-58565-5_10","DOI":"10.1007\/978-3-030-58565-5_10"},{"key":"12","doi-asserted-by":"crossref","unstructured":"[12] W. Ge, X. Lin, and Y. Yu, \u201cWeakly supervised complementary parts models for fine-grained image classification from the bottom up,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.3034-3043, 2019. 10.1109\/cvpr.2019.00315","DOI":"10.1109\/CVPR.2019.00315"},{"key":"13","doi-asserted-by":"crossref","unstructured":"[13] Z. Ge, S. Liu, Z. Li, O. Yoshie, and J. Sun, \u201cOTA: Optimal transport assignment for object detection,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.303-312, 2021. 10.1109\/cvpr46437.2021.00037","DOI":"10.1109\/CVPR46437.2021.00037"},{"key":"14","unstructured":"[14] D. Ha, A. Dai, and Q.V. Le, \u201cHyperNetworks,\u201d arXiv preprint arXiv:1609.09106, 2016. 10.48550\/arXiv.1609.09106"},{"key":"15","doi-asserted-by":"crossref","unstructured":"[15] K. He, X. Zhang, S. Ren, and J. Sun, \u201cDeep residual learning for image recognition,\u201d Proc. IEEE Conference on Computer Vision and Pattern Recognition, pp.770-778, 2016. 10.1109\/cvpr.2016.90","DOI":"10.1109\/CVPR.2016.90"},{"key":"16","unstructured":"[16] A.G. Howard, M. Zhu, B. Chen, D. Kalenichenko, W. Wang, T. Weyand, M. Andreetto, and H. Adam, \u201cMobileNets: Efficient convolutional neural networks for mobile vision applications,\u201d arXiv preprint arXiv:1704.04861, 2017. 10.48550\/arXiv.1704.04861"},{"key":"17","unstructured":"[17] F.N. Iandola, S. Han, M.W. Moskewicz, K. Ashraf, W.J. Dally, and K. Keutzer, \u201cSqueezeNet: AlexNet-level accuracy with 50x fewer parameters and &lt;0.5MB model size,\u201d arXiv preprint arXiv:1602.07360, 2016. 10.48550\/arXiv.1602.07360"},{"key":"18","doi-asserted-by":"crossref","unstructured":"[18] R. Ji, L. Wen, L. Zhang, D. Du, Y. Wu, C. Zhao, X. Liu, and F. Huang, \u201cAttention convolutional binary neural tree for fine-grained visual categorization,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.10468-10477, 2020. 10.1109\/cvpr42600.2020.01048","DOI":"10.1109\/CVPR42600.2020.01048"},{"key":"19","doi-asserted-by":"crossref","unstructured":"[19] K. Kim and H.S. Lee, \u201cProbabilistic anchor assignment with iou prediction for object detection,\u201d European Conference on Computer Vision, pp.355-371, Springer, 2020. 10.1007\/978-3-030-58595-2_22","DOI":"10.1007\/978-3-030-58595-2_22"},{"key":"20","doi-asserted-by":"publisher","unstructured":"[20] T. Kong, F. Sun, H. Liu, Y. Jiang, L. Li, and J. Shi, \u201cFoveaBox: Beyound anchor-based object detection,\u201d IEEE Trans. Image Process., vol.29, pp.7389-7398, 2020. 10.1109\/tip.2020.3002345","DOI":"10.1109\/TIP.2020.3002345"},{"key":"21","unstructured":"[21] S. Kornblith, M. Norouzi, H. Lee, and G. Hinton, \u201cSimilarity of neural network representations revisited,\u201d International Conference on Machine Learning, pp.3519-3529, PMLR, 2019."},{"key":"22","doi-asserted-by":"crossref","unstructured":"[22] J. Krause, M. Stark, J. Deng, and L. Fei-Fei, \u201c3D object representations for fine-grained categorization,\u201d Proc. IEEE International Conference on Computer Vision Workshops, pp.554-561, 2013. 10.1109\/iccvw.2013.77","DOI":"10.1109\/ICCVW.2013.77"},{"key":"23","doi-asserted-by":"crossref","unstructured":"[23] H. Li, Z. Wu, C. Zhu, C. Xiong, R. Socher, and L.S. Davis, \u201cLearning from noisy anchors for one-stage object detection,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.10588-10597, 2020. 10.1109\/cvpr42600.2020.01060","DOI":"10.1109\/CVPR42600.2020.01060"},{"key":"24","doi-asserted-by":"crossref","unstructured":"[24] Y. Li, S. Gu, L.V. Gool, and R. Timofte, \u201cLearning filter basis for convolutional neural network compression,\u201d Proc. IEEE\/CVF International Conference on Computer Vision, pp.5623-5632, 2019. 10.1109\/iccv.2019.00572","DOI":"10.1109\/ICCV.2019.00572"},{"key":"25","doi-asserted-by":"crossref","unstructured":"[25] T.-Y. Lin, P. Doll\u00e1r, R. Girshick, K. He, B. Hariharan, and S. Belongie, \u201cFeature pyramid networks for object detection,\u201d Proc. IEEE Conference on Computer Vision and Pattern Recognition, pp.2117-2125, 2017. 10.1109\/cvpr.2017.106","DOI":"10.1109\/CVPR.2017.106"},{"key":"26","doi-asserted-by":"crossref","unstructured":"[26] T.-Y. Lin, P. Goyal, R. Girshick, K. He, and P. Doll\u00e1r, \u201cFocal loss for dense object detection,\u201d Proc. IEEE International Conference on Computer Vision, pp.2980-2988, 2017. 10.1109\/iccv.2017.324","DOI":"10.1109\/ICCV.2017.324"},{"key":"27","doi-asserted-by":"publisher","unstructured":"[27] M. Liu, C. Zhang, H. Bai, R. Zhang, and Y. Zhao, \u201cCross-part learning for fine-grained image classification,\u201d IEEE Trans. Image Process., vol.31, pp.748-758, 2021. 10.1109\/tip.2021.3135477","DOI":"10.1109\/TIP.2021.3135477"},{"key":"28","doi-asserted-by":"publisher","unstructured":"[28] S. Liu, W. Jiang, L. Wu, H. Wen, M. Liu, and Y. Wang, \u201cReal-time classification of rubber wood boards using an SSR-based CNN,\u201d IEEE Trans. Instrum. Meass, vol.69, no.11, pp.8725-8734, 2020. 10.1109\/tim.2020.3001370","DOI":"10.1109\/TIM.2020.3001370"},{"key":"29","doi-asserted-by":"crossref","unstructured":"[29] W. Liu, S. Liao, W. Ren, W. Hu, and Y. Yu, \u201cHigh-level semantic feature detection: A new perspective for pedestrian detection,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.5187-5196, 2019. 10.1109\/cvpr.2019.00533","DOI":"10.1109\/CVPR.2019.00533"},{"key":"30","doi-asserted-by":"crossref","unstructured":"[30] Z. Liu, Y. Lin, Y. Cao, H. Hu, Y. Wei, Z. Zhang, S. Lin, and B. Guo, \u201cSwin transformer: Hierarchical vision transformer using shifted windows,\u201d Proc. IEEE\/CVF International Conference on Computer Vision, pp.10012-10022, 2021. 10.1109\/iccv48922.2021.00986","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"31","unstructured":"[31] I. Loshchilov and F. Hutter, \u201cDecoupled weight decay regularization,\u201d arXiv preprint arXiv:1711.05101, 2017. 10.48550\/arXiv.1711.05101"},{"key":"32","unstructured":"[32] S. Maji, E. Rahtu, J. Kannala, M. Blaschko, and A. Vedaldi, \u201cFine-grained visual classification of aircraft,\u201d arXiv preprint arXiv:1306.5151, 2013. 10.48550\/arXiv.1306.5151"},{"key":"33","unstructured":"[33] P. Michel, O. Levy, and G. Neubig, \u201cAre sixteen heads really better than one?,\u201d Advances in Neural Information Processing Systems, vol.32, pp.14014-14024, 2019."},{"key":"34","unstructured":"[34] A. Morcos, M. Raghu, and S. Bengio, \u201cInsights on representational similarity in neural networks with canonical correlation,\u201d Advances in Neural Information Processing Systems, 31, 2018."},{"key":"35","doi-asserted-by":"crossref","unstructured":"[35] Z. Peng, W. Huang, S. Gu, L. Xie, Y. Wang, J. Jiao, and Q. Ye, \u201cConformer: Local features coupling global representations for visual recognition,\u201d Proc. IEEE\/CVF International Conference on Computer Vision, pp.367-376, 2021. 10.1109\/iccv48922.2021.00042","DOI":"10.1109\/ICCV48922.2021.00042"},{"key":"36","unstructured":"[36] Q. Qiu, X. Cheng, G. Sapiro, et al., \u201cDCFNet: Deep neural network with decomposed convolutional filters,\u201d International Conference on Machine Learning, pp.4198-4207, PMLR, 2018."},{"key":"37","unstructured":"[37] M. Raghu, J. Gilmer, J. Yosinski, and J. Sohl-Dickstein, \u201cSVCCA: Singular vector canonical correlation analysis for deep learning dynamics and interpretability,\u201d Advances in Neural Information Processing Systems, 30, 2017."},{"key":"38","unstructured":"[38] S. Ren, K. He, R. Girshick, and J. Sun, \u201cFaster R-CNN: Towards real-time object detection with region proposal networks,\u201d Advances in Neural Information Processing Systems, vol.28, pp.91-99, 2015."},{"key":"39","doi-asserted-by":"crossref","unstructured":"[39] F. Santambrogio, Optimal Transport for Applied Mathematicians, Birk\u00e4user, NY, 55(58-63):94, 2015. 10.1007\/978-3-319-20828-2","DOI":"10.1007\/978-3-319-20828-2"},{"key":"40","unstructured":"[40] P. Savarese and M. Maire, \u201cLearning implicitly recurrent CNNs through parameter sharing,\u201d arXiv preprint arXiv:1902.09701, 2019. 10.48550\/arXiv.1902.09701"},{"key":"41","unstructured":"[41] S. Shao, Z. Zhao, B. Li, T. Xiao, G. Yu, X. Zhang, and J. Sun, \u201cCrowdHuman: A benchmark for detecting human in a crowd,\u201d arXiv preprint arXiv:1805.00123, 2018. 10.48550\/arXiv.1805.00123"},{"key":"42","doi-asserted-by":"crossref","unstructured":"[42] S. Son, S. Nah, and K.M. Lee, \u201cClustering convolutional kernels to compress deep neural networks,\u201d Proc. European Conference on Computer Vision (ECCV), pp.216-232, 2018. 10.1007\/978-3-030-01237-3_14","DOI":"10.1007\/978-3-030-01237-3_14"},{"key":"43","unstructured":"[43] I. Sosnovik, M. Szmaja, and A. Smeulders, \u201cScale-equivariant steerable networks,\u201d arXiv preprint arXiv:1910.11093, 2019. 10.48550\/arXiv.1910.11093"},{"key":"44","doi-asserted-by":"crossref","unstructured":"[44] Z. Tian, C. Shen, H. Chen, and T. He, \u201cFCOS: Fully convolutional one-stage object detection,\u201d Proc. IEEE\/CVF International Conference on Computer Vision, pp.9627-9636, 2019. 10.1109\/iccv.2019.00972","DOI":"10.1109\/ICCV.2019.00972"},{"key":"45","unstructured":"[45] A. Vaswani, N. Shazeer, N. Parmar, J. Uszkoreit, L. Jones, A.N. Gomez, \u0141. Kaiser, and I. Polosukhin, \u201cAttention is all you need,\u201d Advances in Neural Information Processing Systems, 30, 2017."},{"key":"46","unstructured":"[46] C. Wah, S. Branson, P. Welinder, P. Perona, and S. Belongie, \u201cThe caltech-ucsd birds-200-2011 dataset,\u201d Technical Report, CNSTR-2011-001, California Inst. Technol., Pasadena, CA, USA, 2011."},{"key":"47","doi-asserted-by":"publisher","unstructured":"[47] H. Wang, Y. Ji, K. Song, M. Sun, P. Lv, and T. Zhang, \u201cViT-P: Classification of genitourinary syndrome of menopause from OCT images based on vision transformer models,\u201d IEEE Trans. Instrum. Meas., vol.70, pp.1-14, 2021. 10.1109\/tim.2021.3122121","DOI":"10.1109\/TIM.2021.3122121"},{"key":"48","doi-asserted-by":"crossref","unstructured":"[48] J. Wang, K. Chen, S. Yang, C.C. Loy, and D. Lin, \u201cRegion proposal by guided anchoring,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.2965-2974, 2019. 10.1109\/cvpr.2019.00308","DOI":"10.1109\/CVPR.2019.00308"},{"key":"49","doi-asserted-by":"crossref","unstructured":"[49] Z. Wang, S. Wang, S. Yang, H. Li, J. Li, and Z. Li, \u201cWeakly supervised fine-grained image classification via Guassian mixture model oriented discriminative learning,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.9749-9758, 2020. 10.1109\/cvpr42600.2020.00977","DOI":"10.1109\/CVPR42600.2020.00977"},{"key":"50","doi-asserted-by":"crossref","unstructured":"[50] A. Wong, M. Famuori, M.J. Shafiee, F. Li, B. Chwyl, and J. Chung, \u201cYOLO Nano: A highly compact you only look once convolutional neural network for object detection,\u201d 2019 Fifth Workshop on Energy Efficient Machine Learning and Cognitive Computing-NeurIPS Edition (EMC2-NIPS), pp.22-25. IEEE, 2019. 10.1109\/emc2-nips53020.2019.00013","DOI":"10.1109\/EMC2-NIPS53020.2019.00013"},{"key":"51","unstructured":"[51] B. Wu, C. Xu, X. Dai, A. Wan, P. Zhang, Z. Yan, M. Tomizuka, J. Gonzalez, K. Keutzer, and P. Vajda, \u201cVisual transformers: Token-based image representation and processing for computer vision,\u201d arXiv preprint arXiv:2006.03677, 2020. 10.48550\/arXiv.2006.03677"},{"key":"52","unstructured":"[52] J. Wu, Y. Wang, Z. Wu, Z. Wang, A. Veeraraghavan, and Y. Lin, \u201cDeep k-means: Re-training and parameter sharing with harder cluster assignments for compressing deep convolutions,\u201d International Conference on Machine Learning, pp.5363-5372, PMLR, 2018."},{"key":"53","unstructured":"[53] T. Xiao, P. Dollar, M. Singh, E. Mintun, T. Darrell, and R. Girshick, \u201cEarly convolutions help transformers see better,\u201d Advances in Neural Information Processing Systems, 34, 2021."},{"key":"54","doi-asserted-by":"crossref","unstructured":"[54] F. Yang, H. Yang, J. Fu, H. Lu, and B. Guo, \u201cLearning texture transformer network for image super-resolution,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.5791-5800, 2020. 10.1109\/cvpr42600.2020.00583","DOI":"10.1109\/CVPR42600.2020.00583"},{"key":"55","unstructured":"[55] T. Yang, X. Zhang, Z. Li, W. Zhang, and J. Sun, \u201cMetaAnchor: Learning to detect objects with customized anchors,\u201d arXiv preprint arXiv:1807.00980, 2018. 10.48550\/arXiv.1807.00980"},{"key":"56","doi-asserted-by":"publisher","unstructured":"[56] T. Ye, J. Zhang, Y. Li, X. Zhang, Z. Zhao, and Z. Li, \u201cCT-Net: An efficient network for low-altitude object detection based on convolution and transformer,\u201d IEEE Trans. Instrum. Meas., vol.71, pp.1-12, 2022. 10.1109\/tim.2022.3165838","DOI":"10.1109\/TIM.2022.3165838"},{"key":"57","doi-asserted-by":"publisher","unstructured":"[57] J. Yu, X. Cheng, and Q. Li, \u201cSurface defect detection of steel strips based on anchor-free network with channel attention and bidirectional feature fusion,\u201d IEEE Trans. Instrum. Meas., vol.71, pp.1-10, 2022. 10.1109\/tim.2021.3136183","DOI":"10.1109\/TIM.2021.3136183"},{"key":"58","doi-asserted-by":"crossref","unstructured":"[58] W. Yu, M. Luo, P. Zhou, C. Si, Y. Zhou, X. Wang, J. Feng, and S. Yan, \u201cMetaFormer is actually what you need for vision,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.10819-10829, 2022. 10.1109\/cvpr52688.2022.01055","DOI":"10.1109\/CVPR52688.2022.01055"},{"key":"59","doi-asserted-by":"crossref","unstructured":"[59] L. Yuan, Y. Chen, T. Wang, W. Yu, Y. Shi, Z.-H. Jiang, F.E. Tay, J. Feng, and S. Yan, \u201cTokens-to-token ViT: Training vision transformers from scratch on imagenet,\u201d Proc. IEEE\/CVF International Conference on Computer Vision, pp.558-567, 2021. 10.1109\/iccv48922.2021.00060","DOI":"10.1109\/ICCV48922.2021.00060"},{"key":"60","doi-asserted-by":"crossref","unstructured":"[60] S. Zagoruyko and N. Komodakis, \u201cWide residual networks,\u201d arXiv preprint arXiv:1605.07146, 2016. 10.48550\/arXiv.1605.07146","DOI":"10.5244\/C.30.87"},{"key":"61","doi-asserted-by":"crossref","unstructured":"[61] S. Zhang, C. Chi, Y. Yao, Z. Lei, and S.Z. Li, \u201cBridging the gap between anchor-based and anchor-free detection via adaptive training sample selection,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.9759-9768, 2020. 10.1109\/cvpr42600.2020.00978","DOI":"10.1109\/CVPR42600.2020.00978"},{"key":"62","unstructured":"[62] X. Zhang, F. Wan, C. Liu, R. Ji, and Q. Ye, \u201cFreeAnchor: Learning to match anchors for visual object detection,\u201d arXiv preprint arXiv:1909.02466, 2019. 10.48550\/arXiv.1909.02466"},{"key":"63","doi-asserted-by":"crossref","unstructured":"[63] X. Zhang, X. Zhou, M. Lin, and J. Sun, \u201cShuffleNet: An extremely efficient convolutional neural network for mobile devices,\u201d Proc. IEEE Conference on Computer Vision and Pattern Recognition, pp.6848-6856, 2018. 10.1109\/cvpr.2018.00716","DOI":"10.1109\/CVPR.2018.00716"},{"key":"64","doi-asserted-by":"crossref","unstructured":"[64] H. Zheng, J. Fu, Z.-J. Zha, and J. Luo, \u201cLooking for the devil in the details: Learning trilinear attention sampling network for fine-grained image recognition,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.5012-5021, 2019. 10.1109\/cvpr.2019.00515","DOI":"10.1109\/CVPR.2019.00515"},{"key":"65","doi-asserted-by":"publisher","unstructured":"[65] H. Zheng, J. Fu, Z.-J. Zha, J. Luo, and T. Mei, \u201cLearning rich part hierarchies with progressive attention networks for fine-grained image recognition,\u201d IEEE Trans. Image Process., vol.29, pp.476-488, 2019. 10.1109\/tip.2019.2921876","DOI":"10.1109\/TIP.2019.2921876"},{"key":"66","doi-asserted-by":"crossref","unstructured":"[66] L. Zhou, Y. Zhou, J.J. Corso, R. Socher, and C. Xiong, \u201cEnd-to-end dense video captioning with masked transformer,\u201d Proc. IEEE Conference on Computer Vision and Pattern Recognition, pp.8739-8748, 2018. 10.1109\/cvpr.2018.00911","DOI":"10.1109\/CVPR.2018.00911"},{"key":"67","unstructured":"[67] B. Zhu, J. Wang, Z. Jiang, F. Zong, S. Liu, Z. Li, and J. Sun, \u201cAutoAssign: Differentiable label assignment for dense object detection,\u201d arXiv preprint arXiv:2007.03496, 2020. 10.48550\/arXiv.2007.03496"},{"key":"68","doi-asserted-by":"crossref","unstructured":"[68] C. Zhu, F. Chen, Z. Shen, and M. Savvides, \u201cSoft anchor-point object detection,\u201d European Conference on Computer Vision, pp.91-107, Springer, 2020. 10.1007\/978-3-030-58545-7_6","DOI":"10.1007\/978-3-030-58545-7_6"},{"key":"69","doi-asserted-by":"crossref","unstructured":"[69] C. Zhu, Y. He, and M. Savvides, \u201cFeature selective anchor-free module for single-shot object detection,\u201d Proc. IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp.840-849, 2019. 10.1109\/cvpr.2019.00093","DOI":"10.1109\/CVPR.2019.00093"}],"container-title":["IEICE Transactions on Fundamentals of Electronics, Communications and Computer Sciences"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transfun\/E107.A\/1\/E107.A_2022EAP1157\/_pdf","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,5,13]],"date-time":"2024-05-13T04:56:10Z","timestamp":1715576170000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.jstage.jst.go.jp\/article\/transfun\/E107.A\/1\/E107.A_2022EAP1157\/_article"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,1,1]]},"references-count":69,"journal-issue":{"issue":"1","published-print":{"date-parts":[[2024]]}},"URL":"https:\/\/doi.org\/10.1587\/transfun.2022eap1157","relation":{},"ISSN":["0916-8508","1745-1337"],"issn-type":[{"value":"0916-8508","type":"print"},{"value":"1745-1337","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,1,1]]},"article-number":"2022EAP1157"}}