{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,5,24]],"date-time":"2025-05-24T04:02:26Z","timestamp":1748059346287,"version":"3.41.0"},"publisher-location":"Cham","reference-count":75,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031925900","type":"print"},{"value":"9783031925917","type":"electronic"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-92591-7_10","type":"book-chapter","created":{"date-parts":[[2025,5,23]],"date-time":"2025-05-23T07:23:44Z","timestamp":1747985024000},"page":"151-168","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Aligning Object Detector Bounding Boxes with\u00a0Human Preference"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5258-8534","authenticated-orcid":false,"given":"Ombretta","family":"Strafforello","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0328-7647","authenticated-orcid":false,"given":"Osman Semih","family":"Kayhan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4691-6586","authenticated-orcid":false,"given":"Oana","family":"Inel","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9954-0685","authenticated-orcid":false,"given":"Klamer","family":"Schutte","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3913-2786","authenticated-orcid":false,"given":"Jan","family":"van Gemert","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,5,12]]},"reference":[{"key":"10_CR1","unstructured":"Amazon mechanical turk. https:\/\/www.mturk.com\/. Accessed 20 Feb 2023"},{"key":"10_CR2","unstructured":"Prolific. https:\/\/www.prolific.co"},{"issue":"4","key":"10_CR3","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/s00138-021-01206-5","volume":"32","author":"D Aswal","year":"2021","unstructured":"Aswal, D., Shukla, P., Nandi, G.C.: Designing effective power law-based loss function for faster and better bounding box regression. Mach. Vis. Appl. 32(4), 1\u201310 (2021). https:\/\/doi.org\/10.1007\/s00138-021-01206-5","journal-title":"Mach. Vis. Appl."},{"key":"10_CR4","doi-asserted-by":"crossref","unstructured":"Ballan, L., Strafforello, O., Schutte, K.: Long-term behaviour recognition in videos with actor-focused region attention. In: VISIGRAPP (5: VISAPP) (2021)","DOI":"10.5220\/0010215803620369"},{"key":"10_CR5","doi-asserted-by":"crossref","unstructured":"Basharat, A., Gritai, A., Shah, M.: Learning object motion patterns for anomaly detection and improved object detection. In: CVPR. IEEE\/CVF (2008)","DOI":"10.1109\/CVPR.2008.4587510"},{"issue":"1\u20132","key":"10_CR6","doi-asserted-by":"publisher","first-page":"21","DOI":"10.1016\/0378-3758(92)90118-C","volume":"29","author":"A Basu","year":"1991","unstructured":"Basu, A., Ebrahimi, N.: Bayesian approach to life testing and reliability estimation using asymmetric loss function. J. Stat. Plan. Inference 29(1\u20132), 21\u201331 (1991)","journal-title":"J. Stat. Plan. Inference"},{"key":"10_CR7","unstructured":"Beal, J., Kim, E., Tzeng, E., Park, D.H., Zhai, A., Kislyuk, D.: Toward Transformer-Based Object Detection (2020)"},{"key":"10_CR8","doi-asserted-by":"crossref","unstructured":"Bolya, D., Zhou, C., Xiao, F., Lee, Y.J.: YOLACT: real-time instance segmentation. In: ICCV. IEEE\/CVF (2019)","DOI":"10.1109\/ICCV.2019.00925"},{"key":"10_CR9","doi-asserted-by":"crossref","unstructured":"Caba\u00a0Heilbron, F., Escorcia, V., Ghanem, B., Carlos\u00a0Niebles, J.: ActivityNet: a large-scale video benchmark for human activity understanding. In: CVPR (2015)","DOI":"10.1109\/CVPR.2015.7298698"},{"key":"10_CR10","doi-asserted-by":"crossref","unstructured":"Cai, Z., Vasconcelos, N.: Cascade R-CNN: delving into high quality object detection. In: CVPR. IEEE\/CVF (2018)","DOI":"10.1109\/CVPR.2018.00644"},{"key":"10_CR11","doi-asserted-by":"crossref","unstructured":"Cao, Z., Hidalgo, G., Simon, T., Wei, S.E., Sheikh, Y.: OpenPose: realtime multi-person 2d pose estimation using part affinity fields. IEEE Trans. Pattern Anal. Mach. Intell. 43(1) (2019)","DOI":"10.1109\/TPAMI.2019.2929257"},{"key":"10_CR12","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"213","DOI":"10.1007\/978-3-030-58452-8_13","volume-title":"Computer Vision \u2013 ECCV 2020","author":"N Carion","year":"2020","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., Zagoruyko, S.: End-to-end object detection with transformers. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12346, pp. 213\u2013229. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58452-8_13"},{"key":"10_CR13","doi-asserted-by":"crossref","unstructured":"Chavali, N., Agrawal, H., Mahendru, A., Batra, D.: Object-proposal evaluation protocol is \u2018gameable\u2019. In: CVPR. IEEE\/CVF (2016)","DOI":"10.1109\/CVPR.2016.97"},{"key":"10_CR14","doi-asserted-by":"crossref","unstructured":"Chen, Y., Cao, Y., Hu, H., Wang, L.: Memory enhanced global-local aggregation for video object detection. In: CVPR. IEEE\/CVF (2020)","DOI":"10.1109\/CVPR42600.2020.01035"},{"key":"10_CR15","doi-asserted-by":"crossref","unstructured":"Cochran, W.G.: The comparison of percentages in matched samples. Biometrika 37(3\/4) (1950)","DOI":"10.2307\/2332378"},{"key":"10_CR16","doi-asserted-by":"crossref","unstructured":"Dai, Z., Cai, B., Lin, Y., Chen, J.: UP-DETR: unsupervised pre-training for object detection with transformers. In: CVPR, pp. 1601\u20131610. IEEE\/CVF (2021)","DOI":"10.1109\/CVPR46437.2021.00165"},{"key":"10_CR17","doi-asserted-by":"crossref","unstructured":"Di\u00a0Salvo, R., Giordano, D., Kavasidis, I.: A crowdsourcing approach to support video annotation. In: Proceedings of the International Workshop on Video and Image Ground Truth in Computer Vision Applications. ACM (2013)","DOI":"10.1145\/2501105.2501113"},{"key":"10_CR18","doi-asserted-by":"crossref","unstructured":"Doshi, K., Yilmaz, Y.: Fast unsupervised anomaly detection in traffic videos. In: CVPRW. IEEE\/CVF (2020)","DOI":"10.1109\/CVPRW50498.2020.00320"},{"key":"10_CR19","doi-asserted-by":"crossref","unstructured":"Duan, K., Bai, S., Xie, L., Qi, H., Huang, Q., Tian, Q.: CenterNet: keypoint triplets for object detection. In: ICCV. IEEE\/CVF (2019)","DOI":"10.1109\/ICCV.2019.00667"},{"key":"10_CR20","doi-asserted-by":"crossref","unstructured":"Everingham, M., Van\u00a0Gool, L., Williams, C.K., Winn, J., Zisserman, A.: The pascal visual object classes (VOC) challenge. IJCV (2010)","DOI":"10.1007\/s11263-009-0275-4"},{"key":"10_CR21","unstructured":"Everingham, M., Van\u00a0Gool, L., Williams, C.K.I., Winn, J., Zisserman, A.: The PASCAL Visual Object Classes Challenge 2012 (VOC2012) Results. http:\/\/www.pascal-network.org\/challenges\/VOC\/voc2012\/workshop\/index.html"},{"key":"10_CR22","unstructured":"Feng, D., et al.: Labels are not perfect: inferring spatial uncertainty in object detection. IEEE Trans. Intell. Transp. Syst. (2021)"},{"key":"10_CR23","doi-asserted-by":"crossref","unstructured":"Girshick, R.: Fast R-CNN. In: ICCV. IEEE (2015)","DOI":"10.1109\/ICCV.2015.169"},{"key":"10_CR24","doi-asserted-by":"crossref","unstructured":"Girshick, R., Donahue, J., Darrell, T., Malik, J.: Region-based convolutional networks for accurate object detection and segmentation. PAMI (2015)","DOI":"10.1109\/TPAMI.2015.2437384"},{"key":"10_CR25","doi-asserted-by":"crossref","unstructured":"Han, L., Wang, P., Yin, Z., Wang, F., Li, H.: Context and structure mining network for video object detection. IJCV (2021)","DOI":"10.1007\/s11263-021-01507-2"},{"key":"10_CR26","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In: CVPR. IEEE\/CVF (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"10_CR27","doi-asserted-by":"crossref","unstructured":"He, K., Gkioxari, G., Doll\u00e1r, P., Girshick, R.: Mask R-CNN. In: ICCV. IEEE\/CVF (2017)","DOI":"10.1109\/ICCV.2017.322"},{"key":"10_CR28","doi-asserted-by":"crossref","unstructured":"He, K., Gkioxari, G., Doll\u00e1r, P., Girshick, R.B.: Mask R-CNN. CoRR abs\/1703.06870 (2017). http:\/\/arxiv.org\/abs\/1703.06870","DOI":"10.1109\/ICCV.2017.322"},{"key":"10_CR29","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"340","DOI":"10.1007\/978-3-642-33712-3_25","volume-title":"Computer Vision \u2013 ECCV 2012","author":"D Hoiem","year":"2012","unstructured":"Hoiem, D., Chodpathumwan, Y., Dai, Q.: Diagnosing error in object detectors. In: Fitzgibbon, A., Lazebnik, S., Perona, P., Sato, Y., Schmid, C. (eds.) ECCV 2012. LNCS, vol. 7574, pp. 340\u2013353. Springer, Heidelberg (2012). https:\/\/doi.org\/10.1007\/978-3-642-33712-3_25"},{"key":"10_CR30","doi-asserted-by":"crossref","unstructured":"Hosang, J., Benenson, R., Doll\u00e1r, P., Schiele, B.: What makes for effective detection proposals? IEEE Trans. Pattern Anal. Mach. Intell. 38(4) (2015)","DOI":"10.1109\/TPAMI.2015.2465908"},{"key":"10_CR31","doi-asserted-by":"crossref","unstructured":"Jiao, L., et al.: New generation deep learning for video object detection: a survey. IEEE Trans. Neural Netw. Learn. Syst. (2021)","DOI":"10.1109\/TNNLS.2021.3053249"},{"key":"10_CR32","doi-asserted-by":"crossref","unstructured":"Kayhan, O.S., Vredebregt, B., van Gemert, J.C.: Hallucination in object detection\u2013a study in visual part verification. In: ICIP. IEEE (2021)","DOI":"10.1109\/ICIP42928.2021.9506670"},{"key":"10_CR33","doi-asserted-by":"crossref","unstructured":"Krishna, R., et\u00a0al.: Visual genome: connecting language and vision using crowdsourced dense image annotations. IJCV (2017)","DOI":"10.1007\/s11263-016-0981-7"},{"key":"10_CR34","doi-asserted-by":"crossref","unstructured":"Kuznetsova, A., et al.: The open images dataset V4: unified image classification, object detection, and visual relationship detection at scale. IJCV (2020)","DOI":"10.1007\/s11263-020-01316-z"},{"key":"10_CR35","doi-asserted-by":"crossref","unstructured":"Law, H., Deng, J.: CornerNet: detecting objects as paired keypoints. In: ECCV. Springer (2018)","DOI":"10.1007\/978-3-030-01264-9_45"},{"key":"10_CR36","doi-asserted-by":"crossref","unstructured":"Li, X., Li, W., Liu, B., Liu, Q., Yu, N.: Object-oriented anomaly detection in surveillance videos. In: 2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE (2018)","DOI":"10.1109\/ICASSP.2018.8461422"},{"key":"10_CR37","doi-asserted-by":"crossref","unstructured":"Li, Y., Trinh, H., Haas, N., Otto, C., Pankanti, S.: Rail component detection, optimization, and assessment for automatic rail track inspection. IEEE Trans. Intell. Transp. Syst. 15(2) (2013)","DOI":"10.1109\/TITS.2013.2287155"},{"key":"10_CR38","doi-asserted-by":"publisher","first-page":"53","DOI":"10.1016\/j.neucom.2019.04.028","volume":"350","author":"Z Li","year":"2019","unstructured":"Li, Z., Dong, M., Wen, S., Hu, X., Zhou, P., Zeng, Z.: CLU-CNNs: object detection for medical images. Neurocomputing 350, 53\u201359 (2019)","journal-title":"Neurocomputing"},{"key":"10_CR39","doi-asserted-by":"crossref","unstructured":"Lin, T., Doll\u00e1r, P., Girshick, R., He, K., Hariharan, B., Belongie, S.: Feature pyramid networks for object detection. In: CVPR. IEEE\/CVF (2017)","DOI":"10.1109\/CVPR.2017.106"},{"key":"10_CR40","doi-asserted-by":"crossref","unstructured":"Lin, T., Goyal, P., Girshick, R., He, K., Doll\u00e1r, P.: Focal loss for dense object detection. In: ICCV. IEEE\/CVF (2017)","DOI":"10.1109\/ICCV.2017.324"},{"key":"10_CR41","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"740","DOI":"10.1007\/978-3-319-10602-1_48","volume-title":"Computer Vision \u2013 ECCV 2014","author":"T-Y Lin","year":"2014","unstructured":"Lin, T.-Y., et al.: Microsoft COCO: common objects in context. In: Fleet, D., Pajdla, T., Schiele, B., Tuytelaars, T. (eds.) ECCV 2014. LNCS, vol. 8693, pp. 740\u2013755. Springer, Cham (2014). https:\/\/doi.org\/10.1007\/978-3-319-10602-1_48"},{"key":"10_CR42","doi-asserted-by":"crossref","unstructured":"Lintott, C.J., et\u00a0al.: Galaxy zoo: morphologies derived from visual inspection of galaxies from the sloan digital sky survey. Monthly Notices of the Royal Astronomical Society (2008)","DOI":"10.1111\/j.1365-2966.2008.13689.x"},{"key":"10_CR43","doi-asserted-by":"crossref","unstructured":"Litjens, G., et al.: A survey on deep learning in medical image analysis. Med. Image Anal. 42 (2017)","DOI":"10.1016\/j.media.2017.07.005"},{"key":"10_CR44","doi-asserted-by":"publisher","unstructured":"Liu, W., et al.: SSD: single shot multibox detector, vol.\u00a09905, pp. 21\u201337 (2016). https:\/\/doi.org\/10.1007\/978-3-319-46448-0_2","DOI":"10.1007\/978-3-319-46448-0_2"},{"key":"10_CR45","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"21","DOI":"10.1007\/978-3-319-46448-0_2","volume-title":"Computer Vision \u2013 ECCV 2016","author":"W Liu","year":"2016","unstructured":"Liu, W., et al.: SSD: single shot multibox detector. In: Leibe, B., Matas, J., Sebe, N., Welling, M. (eds.) ECCV 2016. LNCS, vol. 9905, pp. 21\u201337. Springer, Cham (2016). https:\/\/doi.org\/10.1007\/978-3-319-46448-0_2"},{"key":"10_CR46","doi-asserted-by":"crossref","unstructured":"Mery, D., Katsaggelos, A.K.: A logarithmic x-ray imaging model for baggage inspection: simulation and object detection. In: CVPRW. IEEE\/CVF (2017)","DOI":"10.1109\/CVPRW.2017.37"},{"key":"10_CR47","doi-asserted-by":"crossref","unstructured":"Padilla, R., Netto, S.L., da\u00a0Silva, E.A.: A survey on performance metrics for object-detection algorithms. In: 2020 International Conference on Systems, Signals and Image Processing (IWSSIP). IEEE (2020)","DOI":"10.1109\/IWSSIP48289.2020.9145130"},{"key":"10_CR48","doi-asserted-by":"crossref","unstructured":"Papadopoulos, D.P., Uijlings, J.R.R., Keller, F., Ferrari, V.: We don\u2019t need no bounding-boxes: training object class detectors using only human verification. In: CVPR. IEEE\/CVF (2016)","DOI":"10.1109\/CVPR.2016.99"},{"key":"10_CR49","doi-asserted-by":"crossref","unstructured":"Prasad, D.K., Dong, H., Rajan, D., Quek, C.: Are object detection assessment criteria ready for maritime computer vision? IEEE Trans. Intell. Transp. Syst. 21(12) (2019)","DOI":"10.1109\/TITS.2019.2954464"},{"key":"10_CR50","doi-asserted-by":"crossref","unstructured":"Redmon, J., Divvala, S., Girshick, R., Farhadi, A.: You only look once: unified, real-time object detection. In: CVPR. IEEE\/CVF (2016)","DOI":"10.1109\/CVPR.2016.91"},{"key":"10_CR51","unstructured":"Redmon, J., Farhadi, A.: YOLOv3: an incremental improvement. arXiv preprint arXiv:1804.02767 (2018)"},{"key":"10_CR52","doi-asserted-by":"crossref","unstructured":"Redmon, J., Farhadi, A.: YOLO9000: better, faster, stronger. In: CVPR. IEEE\/CVF (2017)","DOI":"10.1109\/CVPR.2017.690"},{"key":"10_CR53","unstructured":"Ren, S., He, K., Girshick, R., Sun, J.: Faster R-CNN: towards real-time object detection with region proposal networks. In: Advances in Neural Information Processing Systems (2015)"},{"key":"10_CR54","doi-asserted-by":"crossref","unstructured":"Ridnik, T., et al.: Asymmetric loss for multi-label classification. In: CVPR. IEEE\/CVF (2021)","DOI":"10.1109\/ICCV48922.2021.00015"},{"key":"10_CR55","doi-asserted-by":"crossref","unstructured":"Russakovsky, O., et al.: ImageNet large scale visual recognition challenge. IJCV (2015)","DOI":"10.1007\/s11263-015-0816-y"},{"key":"10_CR56","doi-asserted-by":"crossref","unstructured":"Russakovsky, O., Li, L.J., Fei-Fei, L.: Best of both worlds: human-machine collaboration for object annotation. In: CVPR. IEEE\/CVF (2015)","DOI":"10.1109\/CVPR.2015.7298824"},{"key":"10_CR57","unstructured":"Russell, B.C., Torralba, A., Murphy, K.P., Freeman, W.T.: LabelMe: a database and web-based tool for image annotation. IJCV (2008)"},{"key":"10_CR58","volume-title":"Introduction to Modern Information Retrieval","author":"G Salton","year":"1986","unstructured":"Salton, G., McGill, M.J.: Introduction to Modern Information Retrieval. McGraw-Hill Inc, USA (1986)"},{"key":"10_CR59","doi-asserted-by":"crossref","unstructured":"Shen, Y., et al.: Parallel detection-and-segmentation learning for weakly supervised instance segmentation. In: ICCV. IEEE\/CVF (2021)","DOI":"10.1109\/ICCV48922.2021.00809"},{"key":"10_CR60","doi-asserted-by":"crossref","unstructured":"Sobti, A., Mavi, V., Balakrishnan, M., Arora, C.: VmAP: a fair metric for video object detection. In: Proceedings of the 29th ACM International Conference on Multimedia (2021)","DOI":"10.1145\/3474085.3475383"},{"key":"10_CR61","unstructured":"Song, S., Zhang, L., Xiao, J.: Robot in a room: toward perfect object recognition in closed environments. CoRR, abs\/1507.02703 (2015)"},{"key":"10_CR62","doi-asserted-by":"crossref","unstructured":"Strafforello, O., Rajasekar, V., Kayhan, O.S., Inel, O., van Gemert, J.C.: Humans disagree with the IoU for measuring object detector localization error. In: ICIP. IEEE (2022)","DOI":"10.1109\/ICIP46576.2022.9898043"},{"key":"10_CR63","unstructured":"Su, H., Deng, J., Fei-Fei, L.: Crowdsourcing annotations for visual object detection. In: Workshops at the Twenty-Sixth AAAI Conference on Artificial Intelligence (2012)"},{"key":"10_CR64","doi-asserted-by":"crossref","unstructured":"Sun, K., Xiao, B., Liu, D., Wang, J.: Deep high-resolution representation learning for human pose estimation. In: CVPR. IEEE\/CVF (2019)","DOI":"10.1109\/CVPR.2019.00584"},{"key":"10_CR65","doi-asserted-by":"publisher","unstructured":"Tian, D., Han, Y., Wang, S., Chen, X., Guan, T.: Absolute size IoU loss for the bounding box regression of the object detection. Neurocomputing 500, 1029\u20131040 (2022). https:\/\/doi.org\/10.1016\/j.neucom.2022.06.018. https:\/\/www.sciencedirect.com\/science\/article\/pii\/S0925231222007378","DOI":"10.1016\/j.neucom.2022.06.018"},{"key":"10_CR66","doi-asserted-by":"crossref","unstructured":"Vondrick, C., Patterson, D., Ramanan, D.: Efficiently scaling up crowdsourced video annotation. IJCV (2013)","DOI":"10.1007\/s11263-012-0564-1"},{"key":"10_CR67","unstructured":"Weisstein, E.W.: Bonferroni correction (2004). https:\/\/mathworld.wolfram.com\/"},{"key":"10_CR68","unstructured":"Wu, Y., Kirillov, A., Massa, F., Lo, W., Girshick, R.: Detectron2 (2019). https:\/\/github.com\/facebookresearch\/detectron2"},{"key":"10_CR69","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"472","DOI":"10.1007\/978-3-030-01231-1_29","volume-title":"Computer Vision \u2013 ECCV 2018","author":"B Xiao","year":"2018","unstructured":"Xiao, B., Wu, H., Wei, Y.: Simple baselines for human pose estimation and tracking. In: Ferrari, V., Hebert, M., Sminchisescu, C., Weiss, Y. (eds.) ECCV 2018. LNCS, vol. 11210, pp. 472\u2013487. Springer, Cham (2018). https:\/\/doi.org\/10.1007\/978-3-030-01231-1_29"},{"key":"10_CR70","doi-asserted-by":"crossref","unstructured":"Xiao, J., Hays, J., Ehinger, K.A., Oliva, A., Torralba, A.: Sun database: large-scale scene recognition from abbey to zoo. In: 2010 IEEE computer society conference on Computer Vision and Pattern Recognition (2010)","DOI":"10.1109\/CVPR.2010.5539970"},{"key":"10_CR71","doi-asserted-by":"crossref","unstructured":"Yuen, J., Russell, B., Liu, C., Torralba, A.: LabelMe video: building a video database with human annotations. In: CVPR. IEEE\/CVF (2009)","DOI":"10.1109\/ICCV.2009.5459289"},{"key":"10_CR72","doi-asserted-by":"crossref","unstructured":"Zheng, T., Zhao, S., Liu, Y., Liu, Z., Cai, D.: SCALoss: side and corner aligned loss for bounding box regression. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol.\u00a036, pp. 3535\u20133543 (2022)","DOI":"10.1609\/aaai.v36i3.20265"},{"key":"10_CR73","doi-asserted-by":"crossref","unstructured":"Zhou, X., Zhuo, J., Krahenbuhl, P.: Bottom-up object detection by grouping extreme and center points. In: CVPR. IEEE\/CVF (2019)","DOI":"10.1109\/CVPR.2019.00094"},{"key":"10_CR74","doi-asserted-by":"crossref","unstructured":"Zhu, X., Vondrick, C., Ramanan, D., Fowlkes, C.C.: Do we need more training data or better models for object detection?. In: BMVC, vol.\u00a03. Citeseer (2012)","DOI":"10.5244\/C.26.80"},{"key":"10_CR75","unstructured":"Zhu, X., Su, W., Lu, L., Li, B., Wang, X., Dai, J.: Deformable DETR: deformable transformers for end-to-end object detection. arXiv preprint arXiv:2010.04159 (2020)"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024 Workshops"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-92591-7_10","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,5,23]],"date-time":"2025-05-23T07:24:21Z","timestamp":1747985061000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-92591-7_10"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9783031925900","9783031925917"],"references-count":75,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-92591-7_10","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"12 May 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}