{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,17]],"date-time":"2026-04-17T04:28:12Z","timestamp":1776400092665,"version":"3.51.2"},"reference-count":58,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T00:00:00Z","timestamp":1740096000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T00:00:00Z","timestamp":1740096000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"name":"Anhui Provincial Key Research and Development Program","award":["2022a05020042"],"award-info":[{"award-number":["2022a05020042"]}]},{"name":"Natural Science Research Project of Anhui Educational Committee","award":["KJ2020A0651"],"award-info":[{"award-number":["KJ2020A0651"]}]},{"name":"Natural Science Research Project of Anhui Educational Committee","award":["2022AH051783"],"award-info":[{"award-number":["2022AH051783"]}]},{"name":"Hefei Municipal Natural Science Foundation","award":["HZR2447"],"award-info":[{"award-number":["HZR2447"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2025,4]]},"DOI":"10.1007\/s00530-025-01720-w","type":"journal-article","created":{"date-parts":[[2025,2,21]],"date-time":"2025-02-21T14:31:16Z","timestamp":1740148276000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Rwkv-vg: visual grounding with RWKV-driven encoder-decoder framework"],"prefix":"10.1007","volume":"31","author":[{"given":"Fudong","family":"Nian","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yanhong","family":"Gu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wentao","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Aoyu","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Dong","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fanding","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,2,21]]},"reference":[{"issue":"6","key":"1720_CR1","first-page":"1","volume":"19","author":"K Li","year":"2023","unstructured":"Li, K., Li, J., Guo, D., Yang, X., Wang, M.: Transformer-based visual grounding with cross-modality interaction. ACM Trans. Multimed. Comput. Commun. Appl. 19(6), 1\u201319 (2023)","journal-title":"ACM Trans. Multimed. Comput. Commun. Appl."},{"issue":"4","key":"1720_CR2","doi-asserted-by":"publisher","first-page":"2073","DOI":"10.1007\/s00530-023-01097-8","volume":"29","author":"X Xu","year":"2023","unstructured":"Xu, X., Lv, G., Sun, Y., Hu, Y., Nian, F.: Hierarchical cross-modal contextual attention network for visual grounding. Multimedia Syst. 29(4), 2073\u20132083 (2023)","journal-title":"Multimedia Syst."},{"issue":"10","key":"1720_CR3","doi-asserted-by":"publisher","first-page":"12113","DOI":"10.1109\/TPAMI.2023.3275156","volume":"45","author":"P Xu","year":"2023","unstructured":"Xu, P., Zhu, X., Clifton, D.A.: Multimodal learning with transformers: a survey. IEEE Trans. Pattern Anal. Mach. Intell. 45(10), 12113\u201312132 (2023)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"1720_CR4","doi-asserted-by":"crossref","unstructured":"Deng, J., Yang, Z., Chen, T., Zhou, W., Li, H.: Transvg: End-to-end visual grounding with transformers. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 1769\u20131779 (2021)","DOI":"10.1109\/ICCV48922.2021.00179"},{"key":"1720_CR5","doi-asserted-by":"crossref","unstructured":"Ye, J., Tian, J., Yan, M., Yang, X., Wang, X., Zhang, J., He, L., Lin, X.: Shifting more attention to visual backbone: Query-modulated refinement networks for end-to-end visual grounding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15502\u201315512 (2022)","DOI":"10.1109\/CVPR52688.2022.01506"},{"key":"1720_CR6","doi-asserted-by":"crossref","unstructured":"Peng, B., Alcaide, E., Anthony, Q., Albalak, A., Arcadinho, S., Biderman, S., Cao, H., Cheng, X., Chung, M., Grella, M., et al.: Rwkv: Reinventing rnns for the transformer era. (2023). arXiv preprint arXiv:2305.13048","DOI":"10.18653\/v1\/2023.findings-emnlp.936"},{"key":"1720_CR7","unstructured":"Wang, J., Yin, W., Long, X., Zhang, X., Xing, Z., Guo, X., Zhang, Q.: Occrwkv: Rethinking efficient 3d semantic occupancy prediction with linear complexity. (2024). arXiv preprint arXiv:2409.19987"},{"key":"1720_CR8","doi-asserted-by":"publisher","DOI":"10.1016\/j.energy.2024.133068","volume":"309","author":"J Hao","year":"2024","unstructured":"Hao, J., Liu, F., Zhang, W.: Multi-scale rwkv with 2-dimensional temporal convolutional network for short-term photovoltaic power forecasting. Energy 309, 133068 (2024)","journal-title":"Energy"},{"key":"1720_CR9","unstructured":"Chen, Z., Li, C., Xie, X., Dube, P.: Onlysportslm: Optimizing sports-domain language models with sota performance under billion parameters. (2024). arXiv preprint arXiv:2409.00286"},{"key":"1720_CR10","unstructured":"Wang, Y., Wang, S., Zhang, J., Fan, K., Wu, J., Jiang, Z., Liu, Y.: Temporal and interactive modeling for efficient human-human motion generation (2024). arXiv preprint arXiv:2408.17135"},{"key":"1720_CR11","doi-asserted-by":"crossref","unstructured":"Zhou, L., Xiao, Z., Ning, Z.: Rwkv-based encoder-decoder model for code completion. In: 2023 3rd International Conference on Electronic Information Engineering and Computer (EIECT), pp. 425\u2013428 (2023). IEEE","DOI":"10.1109\/EIECT60552.2023.10442108"},{"key":"1720_CR12","unstructured":"Liu, S., Fan, X., Wu, G.: Why perturbing symbolic music is necessary: Fitting the distribution of never-used notes through a joint probabilistic diffusion model. (2024). arXiv preprint arXiv:2408.01950"},{"key":"1720_CR13","doi-asserted-by":"crossref","unstructured":"He, Q., Zhang, J., Peng, J., He, H., Wang, Y., Wang, C.: Pointrwkv: Efficient rwkv-like model for hierarchical point cloud learning. (2024). arXiv preprint arXiv:2405.15214","DOI":"10.1609\/aaai.v39i3.32353"},{"key":"1720_CR14","unstructured":"Fei, Z., Fan, M., Yu, C., Li, D., Huang, J.: Diffusion-rwkv: Scaling rwkv-like architectures for diffusion models. (2024). arXiv preprint arXiv:2404.04478"},{"key":"1720_CR15","doi-asserted-by":"crossref","unstructured":"Hu, R., Rohrbach, M., Andreas, J., Darrell, T., Saenko, K.: Modeling relationships in referential expressions with compositional modular networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1115\u20131124 (2017)","DOI":"10.1109\/CVPR.2017.470"},{"key":"1720_CR16","doi-asserted-by":"crossref","unstructured":"Zhang, H., Niu, Y., Chang, S.-F.: Grounding referring expressions in images by variational context. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4158\u20134166 (2018)","DOI":"10.1109\/CVPR.2018.00437"},{"key":"1720_CR17","doi-asserted-by":"crossref","unstructured":"Zhuang, B., Wu, Q., Shen, C., Reid, I., Van Den\u00a0Hengel, A.: Parallel attention: A unified framework for visual object discovery through dialogs and queries. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 4252\u20134261 (2018)","DOI":"10.1109\/CVPR.2018.00447"},{"key":"1720_CR18","doi-asserted-by":"crossref","unstructured":"Yu, L., Lin, Z., Shen, X., Yang, J., Lu, X., Bansal, M., Berg, T.L.: Mattnet: Modular attention network for referring expression comprehension. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1307\u20131315 (2018)","DOI":"10.1109\/CVPR.2018.00142"},{"key":"1720_CR19","doi-asserted-by":"crossref","unstructured":"Liu, X., Wang, Z., Shao, J., Wang, X., Li, H.: Improving referring expression grounding with cross-modal attention-guided erasing. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 1950\u20131959 (2019)","DOI":"10.1109\/CVPR.2019.00205"},{"issue":"2","key":"1720_CR20","doi-asserted-by":"publisher","first-page":"684","DOI":"10.1109\/TPAMI.2019.2911066","volume":"44","author":"R Hong","year":"2019","unstructured":"Hong, R., Liu, D., Mo, X., He, X., Zhang, H.: Learning to compose and reason with language tree structures for visual grounding. IEEE Trans. Pattern Anal. Mach. Intell. 44(2), 684\u2013696 (2019)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"1720_CR21","doi-asserted-by":"crossref","unstructured":"Liu, D., Zhang, H., Wu, F., Zha, Z.-J.: Learning to assemble neural module tree networks for visual grounding. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 4673\u20134682 (2019)","DOI":"10.1109\/ICCV.2019.00477"},{"issue":"6","key":"1720_CR22","doi-asserted-by":"publisher","first-page":"1137","DOI":"10.1109\/TPAMI.2016.2577031","volume":"39","author":"S Ren","year":"2016","unstructured":"Ren, S., He, K., Girshick, R., Sun, J.: Faster r-cnn: towards real-time object detection with region proposal networks. IEEE Trans. Pattern Anal. Mach. Intell. 39(6), 1137\u20131149 (2016)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"1720_CR23","doi-asserted-by":"crossref","unstructured":"Yang, Z., Gong, B., Wang, L., Huang, W., Yu, D., Luo, J.: A fast and accurate one-stage approach to visual grounding. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 4683\u20134693 (2019)","DOI":"10.1109\/ICCV.2019.00478"},{"key":"1720_CR24","unstructured":"Farhadi, A., Redmon, J.: Yolov3: An incremental improvement. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, vol. 1804, pp. 1\u20136 (2018). Springer Berlin\/Heidelberg, Germany"},{"key":"1720_CR25","doi-asserted-by":"crossref","unstructured":"Yang, Z., Chen, T., Wang, L., Luo, J.: Improving one-stage visual grounding by recursive sub-query construction. In: European Conference on Computer Vision, pp. 387\u2013404 (2020). Springer","DOI":"10.1007\/978-3-030-58568-6_23"},{"key":"1720_CR26","doi-asserted-by":"crossref","unstructured":"Zhu, C., Zhou, Y., Shen, Y., Luo, G., Pan, X., Lin, M., Chen, C., Cao, L., Sun, X., Ji, R.: Seqtr: A simple yet universal network for visual grounding. In: European Conference on Computer Vision, pp. 598\u2013615 (2022). Springer","DOI":"10.1007\/978-3-031-19833-5_35"},{"key":"1720_CR27","doi-asserted-by":"crossref","unstructured":"Ho, C.-H., Appalaraju, S., Jasani, B., Manmatha, R., Vasconcelos, N.: Yoro-lightweight end to end visual grounding. In: European Conference on Computer Vision, pp. 3\u201323 (2022). Springer","DOI":"10.1007\/978-3-031-25085-9_1"},{"key":"1720_CR28","doi-asserted-by":"crossref","unstructured":"Lu, M., Li, R., Feng, F., Ma, Z., Wang, X.: Lgr-net: Language guided reasoning network for referring expression comprehension. IEEE Trans. Circuits Syst. Video Technol. (2024)","DOI":"10.1109\/TCSVT.2024.3374786"},{"key":"1720_CR29","doi-asserted-by":"crossref","unstructured":"Kang, W., Liu, G., Shah, M., Yan, Y.: Segvg: transferring object bounding box to segmentation for visual grounding. In: European Conference on Computer Vision, pp. 57\u201375 (2025). Springer","DOI":"10.1007\/978-3-031-72920-1_4"},{"key":"1720_CR30","doi-asserted-by":"crossref","unstructured":"Xiao, L., Yang, X., Peng, F., Yan, M., Wang, Y., Xu, C.: Clip-vg: Self-paced curriculum adapting of clip for visual grounding. IEEE Trans. Multimedia (2023)","DOI":"10.1109\/TMM.2023.3321501"},{"key":"1720_CR31","unstructured":"Radford, A., Kim, J.W., Hallacy, C., Ramesh, A., Goh, G., Agarwal, S., Sastry, G., Askell, A., Mishkin, P., Clark, J., etal.: Learning transferable visual models from natural language supervision. In: International Conference on Machine Learning, pp. 8748\u20138763 (2021). PMLR"},{"key":"1720_CR32","doi-asserted-by":"crossref","unstructured":"Yang, K., Gu, T., An, X., Jiang, H., Dai, X., Feng, Z., Cai, W., Deng, J.: Clip-cid: Efficient clip distillation via cluster-instance discrimination. (2024). arXiv preprint arXiv:2408.09441","DOI":"10.1609\/aaai.v39i20.35505"},{"key":"1720_CR33","doi-asserted-by":"crossref","unstructured":"An, X., Yang, K., Dai, X., Feng, Z., Deng, J.: Multi-label cluster discrimination for visual representation learning. (2024). arXiv preprint arXiv:2407.17331","DOI":"10.1007\/978-3-031-73383-3_25"},{"key":"1720_CR34","doi-asserted-by":"crossref","unstructured":"Mu, N., Kirillov, A., Wagner, D., Xie, S.: Slip: Self-supervision meets language-image pre-training. In: European Conference on Computer Vision, pp. 529\u2013544 (2022). Springer","DOI":"10.1007\/978-3-031-19809-0_30"},{"key":"1720_CR35","unstructured":"Li, Y., Liang, F., Zhao, L., Cui, Y., Ouyang, W., Shao, J., Yu, F., Yan, J.: Supervision exists everywhere: A data efficient contrastive language-image pre-training paradigm. (2021). arXiv preprint arXiv:2110.05208"},{"key":"1720_CR36","unstructured":"Yao, L., Huang, R., Hou, L., Lu, G., Niu, M., Xu, H., Liang, X., Li, Z., Jiang, X., Xu, C.: Filip: Fine-grained interactive language-image pre-training. (2021). arXiv preprint arXiv:2111.07783"},{"key":"1720_CR37","first-page":"1008","volume":"35","author":"J Lee","year":"2022","unstructured":"Lee, J., Kim, J., Shon, H., Kim, B., Kim, S.H., Lee, H., Kim, J.: Uniclip: Unified framework for contrastive language-image pre-training. Adv. Neural. Inf. Process. Syst. 35, 1008\u20131019 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1720_CR38","unstructured":"Geng, S., Yuan, J., Tian, Y., Chen, Y., Zhang, Y.: Hiclip: Contrastive language-image pretraining with hierarchy-aware attention. (2023). arXiv preprint arXiv:2303.02995"},{"key":"1720_CR39","doi-asserted-by":"crossref","unstructured":"Yang, K., Deng, J., An, X., Li, J., Feng, Z., Guo, J., Yang, J., Liu, T.: Alip: Adaptive language-image pre-training with synthetic caption. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 2922\u20132931 (2023)","DOI":"10.1109\/ICCV51070.2023.00273"},{"key":"1720_CR40","doi-asserted-by":"crossref","unstructured":"Gu, T., Yang, K., An, X., Feng, Z., Liu, D., Cai, W., Deng, J.: Rwkv-clip: A robust vision-language representation learner. (2024). arXiv preprint arXiv:2406.06973","DOI":"10.18653\/v1\/2024.emnlp-main.276"},{"key":"1720_CR41","unstructured":"Peng, B., Goldstein, D., Anthony, Q., Albalak, A., Alcaide, E., Biderman, S., Cheah, E., Ferdinan, T., Hou, H., Kazienko, P., et al.: Eagle and finch: Rwkv with matrix-valued states and dynamic recurrence. (2024). arXiv:2404.05892"},{"key":"1720_CR42","unstructured":"Duan, Y., Wang, W., Chen, Z., Zhu, X., Lu, L., Lu, T., Qiao, Y., Li, H., Dai, J., Wang, W.: Vision-rwkv: Efficient and scalable visual perception with rwkv-like architectures. (2024). arXiv:2403.02308"},{"key":"1720_CR43","unstructured":"Agarap, A.F.: Deep learning using rectified linear units (relu). (2018). arXiv:1803.08375"},{"key":"1720_CR44","doi-asserted-by":"publisher","first-page":"5032","DOI":"10.1109\/TIP.2021.3077144","volume":"30","author":"H Peng","year":"2021","unstructured":"Peng, H., Yu, S.: A systematic iou-related method: Beyond simplified regression for better localization. IEEE Trans. Image Process. 30, 5032\u20135044 (2021)","journal-title":"IEEE Trans. Image Process."},{"key":"1720_CR45","doi-asserted-by":"crossref","unstructured":"Kazemzadeh, S., Ordonez, V., Matten, M., Berg, T.: Referitgame: Referring to objects in photographs of natural scenes. In: Proceedings of the Conference on Empirical Methods in Natural Language Processing (EMNLP), pp. 787\u2013798 (2014)","DOI":"10.3115\/v1\/D14-1086"},{"key":"1720_CR46","doi-asserted-by":"crossref","unstructured":"Yu, L., Poirson, P., Yang, S., Berg, A.C., Berg, T.L.: Modeling context in referring expressions. In: European Conference on Computer Vision, pp. 69\u201385 (2016). Springer","DOI":"10.1007\/978-3-319-46475-6_5"},{"key":"1720_CR47","doi-asserted-by":"crossref","unstructured":"Mao, J., Huang, J., Toshev, A., Camburu, O., Yuille, A.L., Murphy, K.: Generation and comprehension of unambiguous object descriptions. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 11\u201320 (2016)","DOI":"10.1109\/CVPR.2016.9"},{"issue":"4","key":"1720_CR48","doi-asserted-by":"publisher","first-page":"419","DOI":"10.1016\/j.cviu.2009.03.008","volume":"114","author":"HJ Escalante","year":"2010","unstructured":"Escalante, H.J., Hern\u00e1ndez, C.A., Gonzalez, J.A., L\u00f3pez-L\u00f3pez, A., Montes, M., Morales, E.F., Sucar, L.E., Villasenor, L., Grubinger, M.: The segmented and annotated iapr tc-12 benchmark. Comput. Vis. Image Underst. 114(4), 419\u2013428 (2010)","journal-title":"Comput. Vis. Image Underst."},{"key":"1720_CR49","doi-asserted-by":"crossref","unstructured":"Lin, T.-Y., Maire, M., Belongie, S., Hays, J., Perona, P., Ramanan, D., Doll\u00e1r, P., Zitnick, C.L.: Microsoft coco: Common objects in context. In: European Conference on Computer Vision, pp. 740\u2013755 (2014). Springer","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"1720_CR50","doi-asserted-by":"crossref","unstructured":"Chen, L., Ma, W., Xiao, J., Zhang, H., Chang, S.-F.: Ref-nms: Breaking proposal bottlenecks in two-stage referring expression grounding. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 35, pp. 1036\u20131044 (2021)","DOI":"10.1609\/aaai.v35i2.16188"},{"key":"1720_CR51","doi-asserted-by":"crossref","unstructured":"Wang, P., Wu, Q., Cao, J., Shen, C., Gao, L., Hengel, A.v.d.: Neighbourhood watch: Referring expression comprehension via language-guided graph attention networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 1960\u20131968 (2019)","DOI":"10.1109\/CVPR.2019.00206"},{"key":"1720_CR52","doi-asserted-by":"crossref","unstructured":"Liao, Y., Liu, S., Li, G., Wang, F., Chen, Y., Qian, C., Li, B.: A real-time cross-modality correlation filtering method for referring expression comprehension. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 10880\u201310889 (2020)","DOI":"10.1109\/CVPR42600.2020.01089"},{"key":"1720_CR53","doi-asserted-by":"crossref","unstructured":"Ye, J., Lin, X., He, L., Li, D., Chen, Q.: One-stage visual grounding via semantic-aware feature filter. In: Proceedings of the ACM International Conference on Multimedia, pp. 1702\u20131711 (2021)","DOI":"10.1145\/3474085.3475313"},{"key":"1720_CR54","doi-asserted-by":"crossref","unstructured":"Qiu, H., Li, H., Wu, Q., Meng, F., Shi, H., Zhao, T., Ngan, K.N.: Language-aware fine-grained object representation for referring expression comprehension. In: Proceedings of the ACM International Conference on Multimedia, pp. 4171\u20134180 (2020)","DOI":"10.1145\/3394171.3413850"},{"key":"1720_CR55","doi-asserted-by":"crossref","unstructured":"Sun, M., Xiao, J., Lim, E.G.: Iterative shrinking for referring expression grounding using deep reinforcement learning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14060\u201314069 (2021)","DOI":"10.1109\/CVPR46437.2021.01384"},{"key":"1720_CR56","doi-asserted-by":"crossref","unstructured":"Huang, B., Lian, D., Luo, W., Gao, S.: Look before you leap: Learning landmark features for one-stage visual grounding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 16888\u201316897 (2021)","DOI":"10.1109\/CVPR46437.2021.01661"},{"issue":"2","key":"1720_CR57","doi-asserted-by":"publisher","first-page":"394","DOI":"10.1109\/TPAMI.2018.2797921","volume":"41","author":"L Wang","year":"2018","unstructured":"Wang, L., Li, Y., Huang, J., Lazebnik, S.: Learning two-branch neural networks for image-text matching tasks. IEEE Trans. Pattern Anal. Mach. Intell. 41(2), 394\u2013407 (2018)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"1720_CR58","doi-asserted-by":"crossref","unstructured":"Mu, Z., Tang, S., Tan, J., Yu, Q., Zhuang, Y.: Disentangled motif-aware graph learning for phrase grounding. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 35, pp. 13587\u201313594 (2021)","DOI":"10.1609\/aaai.v35i15.17602"}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-025-01720-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-025-01720-w\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-025-01720-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,4,21]],"date-time":"2025-04-21T19:36:41Z","timestamp":1745264201000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-025-01720-w"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,2,21]]},"references-count":58,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2025,4]]}},"alternative-id":["1720"],"URL":"https:\/\/doi.org\/10.1007\/s00530-025-01720-w","relation":{},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"value":"0942-4962","type":"print"},{"value":"1432-1882","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,2,21]]},"assertion":[{"value":"3 December 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"7 February 2025","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 February 2025","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}],"article-number":"124"}}