{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,25]],"date-time":"2025-03-25T19:24:21Z","timestamp":1742930661344,"version":"3.40.3"},"publisher-location":"Cham","reference-count":23,"publisher":"Springer International Publishing","isbn-type":[{"type":"print","value":"9783030863821"},{"type":"electronic","value":"9783030863838"}],"license":[{"start":{"date-parts":[[2021,1,1]],"date-time":"2021-01-01T00:00:00Z","timestamp":1609459200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"},{"start":{"date-parts":[[2021,1,1]],"date-time":"2021-01-01T00:00:00Z","timestamp":1609459200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springer.com\/tdm"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021]]},"DOI":"10.1007\/978-3-030-86383-8_54","type":"book-chapter","created":{"date-parts":[[2021,9,10]],"date-time":"2021-09-10T08:02:49Z","timestamp":1631260969000},"page":"677-689","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Modeling Context-Guided Visual and Linguistic Semantic Feature for Video Captioning"],"prefix":"10.1007","author":[{"given":"Zhixin","family":"Sun","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xian","family":"Zhong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shuqin","family":"Chen","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wenxuan","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Duxiu","family":"Feng","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lin","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2021,9,7]]},"reference":[{"key":"54_CR1","doi-asserted-by":"crossref","unstructured":"Zhang, J., Peng, Y.: Object-aware aggregation with bidirectional temporal graph for video captioning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 8327\u20138336 (2019)","DOI":"10.1109\/CVPR.2019.00852"},{"key":"54_CR2","doi-asserted-by":"crossref","unstructured":"Chen, J., Chao, H.: VideoTRM: pre-training for video captioning challenge 2020. In: Proceedings of the 28th ACM International Conference on Multimedia (MM), pp. 4605\u20134609 (2020)","DOI":"10.1145\/3394171.3416291"},{"key":"54_CR3","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"300","DOI":"10.1007\/978-3-030-30490-4_25","volume-title":"Artificial Neural Networks and Machine Learning \u2013 ICANN 2019: Text and Time Series","author":"T Karayil","year":"2019","unstructured":"Karayil, T., Irfan, A., Raue, F., Hees, J., Dengel, A.: Conditional GANs for image captioning with sentiments. In: Tetko, I.V., K\u016frkov\u00e1, V., Karpov, P., Theis, F. (eds.) ICANN 2019. LNCS, vol. 11730, pp. 300\u2013312. Springer, Cham (2019). https:\/\/doi.org\/10.1007\/978-3-030-30490-4_25"},{"key":"54_CR4","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"281","DOI":"10.1007\/978-3-030-30490-4_23","volume-title":"Artificial Neural Networks and Machine Learning \u2013 ICANN 2019: Text and Time Series","author":"Y Fan","year":"2019","unstructured":"Fan, Y., Xu, J., Sun, Y., Wang, Y.: A novel image captioning method based on generative adversarial networks. In: Tetko, I.V., K\u016frkov\u00e1, V., Karpov, P., Theis, F. (eds.) ICANN 2019. LNCS, vol. 11730, pp. 281\u2013292. Springer, Cham (2019). https:\/\/doi.org\/10.1007\/978-3-030-30490-4_23"},{"key":"54_CR5","doi-asserted-by":"crossref","unstructured":"Chen, J., Pan, Y., Li, Y., Yao, T., Chao, H., Mei, T.: Temporal deformable convolutional encoder-decoder networks for video captioning. In: Proceedings of Conference on Artificial Intelligence (AAAI), pp. 8167\u20138174 (2019)","DOI":"10.1609\/aaai.v33i01.33018167"},{"key":"54_CR6","doi-asserted-by":"crossref","unstructured":"Aafaq, N., Akhtar, N., Liu, W., Gilani, S.Z., Mian, A.: Spatio-temporal dynamics and semantic attribute enriched visual encoding for video captioning. In: Proceedings of IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 12487\u201312496 (2019)","DOI":"10.1109\/CVPR.2019.01277"},{"key":"54_CR7","doi-asserted-by":"crossref","unstructured":"Xu, J., Yao, T., Zhang, Y., Mei, T.: Learning multimodal attention LSTM networks for video captioning. In: Proceedings of ACM International Conference on Multimedia (MM), pp. 537\u2013545 (2017)","DOI":"10.1145\/3123266.3123448"},{"key":"54_CR8","doi-asserted-by":"crossref","unstructured":"Zhang, Z., et al.: Object relational graph with teacher-recommended learning for video captioning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 13278\u201313288 (2020)","DOI":"10.1109\/CVPR42600.2020.01329"},{"key":"54_CR9","doi-asserted-by":"crossref","unstructured":"Wang, B., Ma, L., Zhang, W., Jiang, W., Wang, J., Liu, W.: Controllable video captioning with POS sequence guidance based on gated fusion network. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV), pp. 2641\u20132650 (2019)","DOI":"10.1109\/ICCV.2019.00273"},{"key":"54_CR10","doi-asserted-by":"crossref","unstructured":"Chen, S., Jiang, Y.G.: Motion guided spatial attention for video captioning. In: Proceedings of the AAAI Conference on Artificial Intelligence (AAAI), pp. 8191\u20138198 (2019)","DOI":"10.1609\/aaai.v33i01.33018191"},{"key":"54_CR11","doi-asserted-by":"publisher","first-page":"2353","DOI":"10.1007\/s11063-020-10352-2","volume":"52","author":"S Chen","year":"2020","unstructured":"Chen, S., Zhong, X., Li, L., Liu, W., Gu, C., Zhong, L.: Adaptively converting auxiliary attributes and textual embedding for video captioning based on BiLSTM. Neural Process. Lett. 52, 2353\u20132369 (2020)","journal-title":"Neural Process. Lett."},{"key":"54_CR12","doi-asserted-by":"crossref","unstructured":"Pei, W., Zhang, J., Wang, X., Ke, L., Shen, X., Tai, Y.W.: Memory-attended recurrent network for video captioning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 8347\u20138356 (2019)","DOI":"10.1109\/CVPR.2019.00854"},{"key":"54_CR13","first-page":"1112","volume":"42","author":"L Gao","year":"2019","unstructured":"Gao, L., Li, X., Song, J., Shen, H.T.: Hierarchical LSTMs with adaptive attention for visual captioning. IEEE Trans. Pattern Anal. Mach. Intell. 42, 1112\u20131131 (2019)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"54_CR14","doi-asserted-by":"crossref","unstructured":"Zheng, Q., Wang, C., Tao, D.: Syntax-aware action targeting for video captioning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 13096\u201313105 (2020)","DOI":"10.1109\/CVPR42600.2020.01311"},{"key":"54_CR15","doi-asserted-by":"crossref","unstructured":"Hou, J., Wu, X., Zhang, X., Qi, Y., Jia, Y., Luo, J.: Joint commonsense and relation reasoning for image and video captioning. In: Proceedings of the AAAI Conference on Artificial Intelligence (AAAI), vol. 34, no. 07, pp. 10973\u201310980 (2020)","DOI":"10.1609\/aaai.v34i07.6731"},{"key":"54_CR16","doi-asserted-by":"crossref","unstructured":"Pan, B., et al.: Spatio-temporal graph for video captioning with knowledge distillation. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR), pp. 10870\u201310879 (2020)","DOI":"10.1109\/CVPR42600.2020.01088"},{"key":"54_CR17","doi-asserted-by":"crossref","unstructured":"Tan, G., Liu, D., Wang, M., Zha, Z.J.: Learning to discretely compose reasoning module networks for video captioning. In: Proceedings of International Joint Conference on Artificial Intelligence (IJCAI) (2020)","DOI":"10.24963\/ijcai.2020\/104"},{"key":"54_CR18","doi-asserted-by":"crossref","unstructured":"Zhu, Y., Jiang, S.: Attention-based densely connected LSTM for video captioning. In: Proceedings of the 27th ACM International Conference on Multimedia (MM), pp. 802\u2013810 (2019)","DOI":"10.1145\/3343031.3350932"},{"issue":"12","key":"54_CR19","doi-asserted-by":"publisher","first-page":"3088","DOI":"10.1109\/TPAMI.2019.2920899","volume":"42","author":"W Zhang","year":"2019","unstructured":"Zhang, W., Wang, B., Ma, L., Liu, W.: Reconstruct and represent video contents for captioning via reinforcement learning. IEEE Trans. Pattern Anal. Mach. Intell. 42(12), 3088\u20133101 (2019)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"54_CR20","doi-asserted-by":"crossref","unstructured":"Anderson, P., et al.: Bottom-up and top-down attention for image captioning and visual question answering. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 6077\u20136086 (2018)","DOI":"10.1109\/CVPR.2018.00636"},{"key":"54_CR21","doi-asserted-by":"crossref","unstructured":"Szegedy, C., Ioffe, S., Vanhoucke, V., Alemi, A.: Inception-v4, inception-ResNet and the impact of residual connections on learning. In: Proceedings of Conference on Artificial Intelligence (AAAI), vol. 31, no. 1 (2017)","DOI":"10.1609\/aaai.v31i1.11231"},{"key":"54_CR22","doi-asserted-by":"crossref","unstructured":"Carreira, J., Zisserman, A.: Quo vadis, action recognition? A new model and the kinetics dataset. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR), pp. 6299\u20136308 (2017)","DOI":"10.1109\/CVPR.2017.502"},{"issue":"6","key":"54_CR23","doi-asserted-by":"publisher","first-page":"1137","DOI":"10.1109\/TPAMI.2016.2577031","volume":"39","author":"S Ren","year":"2016","unstructured":"Ren, S., He, K., Girshick, R., Sun, J.: Faster R-CNN: towards real-time object detection with region proposal networks. IEEE Trans. Pattern Anal. Mach. Intell. 39(6), 1137\u20131149 (2016)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."}],"container-title":["Lecture Notes in Computer Science","Artificial Neural Networks and Machine Learning \u2013 ICANN 2021"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-030-86383-8_54","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,3,7]],"date-time":"2024-03-07T14:49:19Z","timestamp":1709822959000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-030-86383-8_54"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021]]},"ISBN":["9783030863821","9783030863838"],"references-count":23,"URL":"https:\/\/doi.org\/10.1007\/978-3-030-86383-8_54","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2021]]},"assertion":[{"value":"7 September 2021","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICANN","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Artificial Neural Networks","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Bratislava","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Slovakia","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2021","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"14 September 2021","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"17 September 2021","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"30","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icann2021","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/e-nns.org\/icann2021\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Single-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"OCS","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"496","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"265","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"4","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"53% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"2.5","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Conference was held online due to the COVID-19 pandemic.","order":10,"name":"additional_info_on_review_process","label":"Additional Info on Review Process","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}