{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,25]],"date-time":"2025-03-25T14:37:11Z","timestamp":1742913431604,"version":"3.40.3"},"publisher-location":"Cham","reference-count":40,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031783975"},{"type":"electronic","value":"9783031783982"}],"license":[{"start":{"date-parts":[[2024,12,2]],"date-time":"2024-12-02T00:00:00Z","timestamp":1733097600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,12,2]],"date-time":"2024-12-02T00:00:00Z","timestamp":1733097600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-78398-2_3","type":"book-chapter","created":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T14:59:50Z","timestamp":1733065190000},"page":"39-55","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Box2Flow: Instance-Based Action Flow Graphs from Videos"],"prefix":"10.1007","author":[{"given":"Jiatong","family":"Li","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kalliopi","family":"Basioti","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Vladimir","family":"Pavlovic","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,12,2]]},"reference":[{"key":"3_CR1","doi-asserted-by":"crossref","unstructured":"Anderson, P., He, X., Buehler, C., Teney, D., Johnson, M., Gould, S., Zhang, L.: Bottom-up and top-down attention for image captioning and visual question answering. In: CVPR (2018)","DOI":"10.1109\/CVPR.2018.00636"},{"key":"3_CR2","unstructured":"Bar, A., Herzig, R., Wang, X., Rohrbach, A., Chechik, G., Darrell, T., Globerson, A.: Compositional video synthesis with action graphs. arXiv preprint arXiv:2006.15327 (2020)"},{"key":"3_CR3","doi-asserted-by":"publisher","first-page":"141","DOI":"10.32620\/reks.2022.1.11","volume":"1","author":"K Bobrovnikova","year":"2022","unstructured":"Bobrovnikova, K., Lysenko, S., Savenko, B., Gaj, P., Savenko, O.: Technique for iot malware detection based on control flow graph analysis. Radioelectronic and Computer Systems 1, 141\u2013153 (2022)","journal-title":"Radioelectronic and Computer Systems"},{"key":"3_CR4","doi-asserted-by":"crossref","unstructured":"Brooks, T., Holynski, A., Efros, A.A.: Instructpix2pix: Learning to follow image editing instructions. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 18392\u201318402 (2023)","DOI":"10.1109\/CVPR52729.2023.01764"},{"issue":"3\u20134","key":"3_CR5","doi-asserted-by":"publisher","first-page":"255","DOI":"10.1016\/S0167-8655(97)00179-7","volume":"19","author":"H Bunke","year":"1998","unstructured":"Bunke, H., Shearer, K.: A graph distance metric based on the maximal common subgraph. Pattern Recogn. Lett. 19(3\u20134), 255\u2013259 (1998)","journal-title":"Pattern Recogn. Lett."},{"key":"3_CR6","doi-asserted-by":"crossref","unstructured":"Cong, Y., Liao, W., Ackermann, H., Rosenhahn, B., Yang, M.Y.: Spatial-temporal transformer for dynamic scene graph generation. In: Proceedings of the IEEE\/CVF international conference on computer vision. pp. 16372\u201316382 (2021)","DOI":"10.1109\/ICCV48922.2021.01606"},{"key":"3_CR7","doi-asserted-by":"crossref","unstructured":"Cong, Y., Yi, J., Rosenhahn, B., Yang, M.Y.: Ssgvs: Semantic scene graph-to-video synthesis. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 2554\u20132564 (2023)","DOI":"10.1109\/CVPRW59228.2023.00254"},{"key":"3_CR8","doi-asserted-by":"crossref","unstructured":"Dvornik, N., Hadji, I., Pham, H., Bhatt, D., Martinez, B., Fazly, A., Jepson, A.D.: Graph2vid: Flow graph to video grounding forweakly-supervised multi-step localization. arXiv preprint arXiv:2210.04996 (2022)","DOI":"10.1007\/978-3-031-19833-5_19"},{"key":"3_CR9","doi-asserted-by":"crossref","unstructured":"Feichtenhofer, C., Fan, H., Malik, J., He, K.: Slowfast networks for video recognition. In: Proceedings of the IEEE\/CVF international conference on computer vision. pp. 6202\u20136211 (2019)","DOI":"10.1109\/ICCV.2019.00630"},{"key":"3_CR10","doi-asserted-by":"crossref","unstructured":"Holm, F., Ghazaei, G., Czempiel, T., \u00d6zsoy, E., Saur, S., Navab, N.: Dynamic scene graph representation for surgical video. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 81\u201387 (2023)","DOI":"10.1109\/ICCVW60793.2023.00015"},{"key":"3_CR11","unstructured":"Houlsby, N., Giurgiu, A., Jastrzebski, S., Morrone, B., De\u00a0Laroussilhe, Q., Gesmundo, A., Attariyan, M., Gelly, S.: Parameter-efficient transfer learning for nlp. In: International conference on machine learning. pp. 2790\u20132799. PMLR (2019)"},{"key":"3_CR12","doi-asserted-by":"crossref","unstructured":"Huang, D.A., Nair, S., Xu, D., Zhu, Y., Garg, A., Fei-Fei, L., Savarese, S., Niebles, J.C.: Neural task graphs: Generalizing to unseen tasks from a single video demonstration. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. pp. 8565\u20138574 (2019)","DOI":"10.1109\/CVPR.2019.00876"},{"key":"3_CR13","doi-asserted-by":"crossref","unstructured":"Huang, Y., Sugano, Y., Sato, Y.: Improving action segmentation via graph-based temporal reasoning. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. pp. 14024\u201314034 (2020)","DOI":"10.1109\/CVPR42600.2020.01404"},{"key":"3_CR14","unstructured":"Hussein, N., Gavves, E., Smeulders, A.W.: Videograph: Recognizing minutes-long human activities in videos. arXiv preprint arXiv:1905.05143 (2019)"},{"key":"3_CR15","doi-asserted-by":"crossref","unstructured":"Jang, Y., Sohn, S., Logeswaran, L., Luo, T., Lee, M., Lee, H.: Multimodal subtask graph generation from instructional videos. arXiv preprint arXiv:2302.08672 (2023)","DOI":"10.18653\/v1\/2023.findings-acl.210"},{"key":"3_CR16","doi-asserted-by":"crossref","unstructured":"Kirillov, A., Mintun, E., Ravi, N., Mao, H., Rolland, C., Gustafson, L., Xiao, T., Whitehead, S., Berg, A.C., Lo, W.Y., et\u00a0al.: Segment anything. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision. pp. 4015\u20134026 (2023)","DOI":"10.1109\/ICCV51070.2023.00371"},{"key":"3_CR17","doi-asserted-by":"crossref","unstructured":"Lei, J., Wang, L., Shen, Y., Yu, D., Berg, T.L., Bansal, M.: Mart: Memory-augmented recurrent transformer for coherent video paragraph captioning. arXiv preprint arXiv:2005.05402 (2020)","DOI":"10.18653\/v1\/2020.acl-main.233"},{"key":"3_CR18","doi-asserted-by":"crossref","unstructured":"Li, Y., Yang, X., Xu, C.: Dynamic scene graph generation via anticipatory pre-training. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 13874\u201313883 (2022)","DOI":"10.1109\/CVPR52688.2022.01350"},{"key":"3_CR19","doi-asserted-by":"crossref","unstructured":"Lin, T.Y., Maire, M., Belongie, S., Hays, J., Perona, P., Ramanan, D., Doll\u00e1r, P., Zitnick, C.L.: Microsoft coco: Common objects in context. In: Computer Vision\u2013ECCV 2014: 13th European Conference, Zurich, Switzerland, September 6-12, 2014, Proceedings, Part V 13. pp. 740\u2013755. Springer (2014)","DOI":"10.1007\/978-3-319-10602-1_48"},{"key":"3_CR20","doi-asserted-by":"crossref","unstructured":"Luo, R., Zhu, Q., Chen, Q., Wang, S., Wei, Z., Sun, W., Tang, S.: Operation diagnosis on procedure graph: The task and dataset. In: Proceedings of the 30th ACM International Conference on Information & Knowledge Management. pp. 3288\u20133292 (2021)","DOI":"10.1145\/3459637.3482157"},{"key":"3_CR21","unstructured":"Mao, W., Desai, R., Iuzzolino, M.L., Kamra, N.: Action dynamics task graphs for learning plannable representations of procedural tasks. arXiv preprint arXiv:2302.05330 (2023)"},{"key":"3_CR22","unstructured":"Mori, S., Maeta, H., Yamakata, Y., Sasada, T.: Flow graph corpus from recipe texts. In: LREC. pp. 2370\u20132377 (2014)"},{"key":"3_CR23","doi-asserted-by":"publisher","first-page":"2125","DOI":"10.1109\/ACCESS.2020.3043452","volume":"9","author":"T Nishimura","year":"2020","unstructured":"Nishimura, T., Hashimoto, A., Ushiku, Y., Kameko, H., Yamakata, Y., Mori, S.: Structure-aware procedural text generation from an image sequence. IEEE Access 9, 2125\u20132141 (2020)","journal-title":"IEEE Access"},{"key":"3_CR24","doi-asserted-by":"crossref","unstructured":"Ost, J., Mannan, F., Thuerey, N., Knodt, J., Heide, F.: Neural scene graphs for dynamic scenes. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 2856\u20132865 (2021)","DOI":"10.1109\/CVPR46437.2021.00288"},{"key":"3_CR25","doi-asserted-by":"crossref","unstructured":"Ou, Y., Mi, L., Chen, Z.: Object-relation reasoning graph for action recognition. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 20133\u201320142 (2022)","DOI":"10.1109\/CVPR52688.2022.01950"},{"key":"3_CR26","doi-asserted-by":"crossref","unstructured":"Pan, L.M., Chen, J., Wu, J., Liu, S., Ngo, C.W., Kan, M.Y., Jiang, Y., Chua, T.S.: Multi-modal cooking workflow construction for food recipes. In: Proceedings of the 28th ACM International Conference on Multimedia. pp. 1132\u20131141 (2020)","DOI":"10.1145\/3394171.3413765"},{"key":"3_CR27","doi-asserted-by":"publisher","first-page":"4491","DOI":"10.1109\/TMM.2020.3042706","volume":"23","author":"L Pan","year":"2020","unstructured":"Pan, L., Chen, J., Liu, S., Ngo, C.W., Kan, M.Y., Chua, T.S.: A hybrid approach for detecting prerequisite relations in multi-modal food recipes. IEEE Trans. Multimedia 23, 4491\u20134501 (2020)","journal-title":"IEEE Trans. Multimedia"},{"key":"3_CR28","doi-asserted-by":"crossref","unstructured":"Rodriguez-Opazo, C., Marrese-Taylor, E., Fernando, B., Li, H., Gould, S.: Dori: Discovering object relationships for moment localization of a natural language query in a video. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision. pp. 1079\u20131088 (2021)","DOI":"10.1109\/WACV48630.2021.00112"},{"key":"3_CR29","doi-asserted-by":"crossref","unstructured":"Schiappa, M.C., Rawat, Y.S.: Svgraph: Learning semantic graphs from instructional videos. In: 2022 IEEE Eighth International Conference on Multimedia Big Data (BigMM). pp. 45\u201352. IEEE (2022)","DOI":"10.1109\/BigMM55396.2022.00014"},{"key":"3_CR30","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2022.109204","volume":"136","author":"Y Tu","year":"2023","unstructured":"Tu, Y., Zhou, C., Guo, J., Li, H., Gao, S., Yu, Z.: Relation-aware attention for video captioning via graph learning. Pattern Recogn. 136, 109204 (2023)","journal-title":"Pattern Recogn."},{"key":"3_CR31","doi-asserted-by":"crossref","unstructured":"Wu, S.C., Wald, J., Tateno, K., Navab, N., Tombari, F.: Scenegraphfusion: Incremental 3d scene graph prediction from rgb-d sequences. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 7515\u20137525 (2021)","DOI":"10.1109\/CVPR46437.2021.00743"},{"key":"3_CR32","unstructured":"Wu, Y., Kirillov, A., Massa, F., Lo, W.Y., Girshick, R.: Detectron2. https:\/\/github.com\/facebookresearch\/detectron2 (2019)"},{"key":"3_CR33","doi-asserted-by":"crossref","unstructured":"Xu, M., Zhao, C., Rojas, D.S., Thabet, A., Ghanem, B.: G-tad: Sub-graph localization for temporal action detection. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. pp. 10156\u201310165 (2020)","DOI":"10.1109\/CVPR42600.2020.01017"},{"key":"3_CR34","unstructured":"Yamakata, Y., Mori, S., Carroll, J.A.: English recipe flow graph corpus. In: Proceedings of the Twelfth Language Resources and Evaluation Conference. pp. 5187\u20135194 (2020)"},{"key":"3_CR35","doi-asserted-by":"crossref","unstructured":"Yamazaki, K., Vo, K., Truong, Q.S., Raj, B., Le, N.: Vltint: Visual-linguistic transformer-in-transformer for coherent video paragraph captioning. In: Proceedings of the AAAI Conference on Artificial intelligence. vol.\u00a037, pp. 3081\u20133090 (2023)","DOI":"10.1609\/aaai.v37i3.25412"},{"key":"3_CR36","doi-asserted-by":"crossref","unstructured":"Zhang, Y., Yamakata, Y., Tajima, K.: Miais: a multimedia recipe dataset with ingredient annotation at each instructional step. In: Proceedings of the 1st International Workshop on Multimedia for Cooking, Eating, and related APPlications. pp. 49\u201352 (2022)","DOI":"10.1145\/3552485.3554938"},{"key":"3_CR37","doi-asserted-by":"crossref","unstructured":"Zhou, H., Mart\u00edn-Mart\u00edn, R., Kapadia, M., Savarese, S., Niebles, J.C.: Procedure-aware pretraining for instructional video understanding. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 10727\u201310738 (2023)","DOI":"10.1109\/CVPR52729.2023.01033"},{"key":"3_CR38","doi-asserted-by":"crossref","unstructured":"Zhou, L., Xu, C., Corso, J.J.: Towards automatic learning of procedures from web instructional videos. In: AAAI Conference on Artificial Intelligence. pp. 7590\u20137598 (2018), https:\/\/www.aaai.org\/ocs\/index.php\/AAAI\/AAAI18\/paper\/view\/17344","DOI":"10.1609\/aaai.v32i1.12342"},{"key":"3_CR39","doi-asserted-by":"crossref","unstructured":"Zhou, L., Zhou, Y., Corso, J.J., Socher, R., Xiong, C.: End-to-end dense video captioning with masked transformer. In: Proceedings of the IEEE conference on computer vision and pattern recognition. pp. 8739\u20138748 (2018)","DOI":"10.1109\/CVPR.2018.00911"},{"key":"3_CR40","doi-asserted-by":"crossref","unstructured":"Zhukov, D., Alayrac, J.B., Cinbis, R.G., Fouhey, D., Laptev, I., Sivic, J.: Cross-task weakly supervised learning from instructional videos. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. pp. 3537\u20133545 (2019)","DOI":"10.1109\/CVPR.2019.00365"}],"container-title":["Lecture Notes in Computer Science","Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-78398-2_3","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,1]],"date-time":"2024-12-01T15:04:17Z","timestamp":1733065457000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-78398-2_3"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,2]]},"ISBN":["9783031783975","9783031783982"],"references-count":40,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-78398-2_3","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2024,12,2]]},"assertion":[{"value":"2 December 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICPR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Pattern Recognition","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Kolkata","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"India","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"1 December 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"5 December 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"27","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icpr2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/icpr2024.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}