{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,12]],"date-time":"2025-12-12T00:06:53Z","timestamp":1765498013741,"version":"3.48.0"},"publisher-location":"New York, NY, USA","reference-count":33,"publisher":"ACM","funder":[{"name":"Shandong Provincial Natural Science Foundation General Project","award":["ZR2023MG069"],"award-info":[{"award-number":["ZR2023MG069"]}]},{"name":"Shandong Province Scienceand Technology Small and Medium-sized Enterprise Innovation Capacity Enhancement Project","award":["2023TSGCO212"],"award-info":[{"award-number":["2023TSGCO212"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,11,10]]},"DOI":"10.1145\/3746252.3761235","type":"proceedings-article","created":{"date-parts":[[2025,11,7]],"date-time":"2025-11-07T23:55:33Z","timestamp":1762559733000},"page":"3103-3112","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["MGFSG-EE: A Method based on Multi-grained Fusion and Scene Graph Enhancement for Event Extraction"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-6874-4114","authenticated-orcid":false,"given":"Xiaoyu","family":"Wang","sequence":"first","affiliation":[{"name":"Qilu University of Technology (Shandong Academy of Sciences), Jinan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2220-930X","authenticated-orcid":false,"given":"Tao","family":"Sun","sequence":"additional","affiliation":[{"name":"Qilu University of Technology (Shandong Academy of Sciences), Jinan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-5571-002X","authenticated-orcid":false,"given":"Gengchen","family":"Liu","sequence":"additional","affiliation":[{"name":"Qilu University of Technology (Shandong Academy of Sciences), Jinan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-3386-1193","authenticated-orcid":false,"given":"Zhi","family":"Yang","sequence":"additional","affiliation":[{"name":"Qilu University of Technology (Shandong Academy of Sciences), Jinan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-3730-6518","authenticated-orcid":false,"given":"Jiahui","family":"Liu","sequence":"additional","affiliation":[{"name":"Qilu University of Technology (Shandong Academy of Sciences), Jinan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-9189-8748","authenticated-orcid":false,"given":"Zimeng","family":"Xu","sequence":"additional","affiliation":[{"name":"Qilu University of Technology (Shandong Academy of Sciences), Jinan, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,11,10]]},"reference":[{"doi-asserted-by":"publisher","key":"e_1_3_2_1_1_1","DOI":"10.1109\/PRML56267.2022.9882203"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_2_1","DOI":"10.3115\/1629235.1629236"},{"key":"e_1_3_2_1_3_1","first-page":"303","volume-title":"Proceedings of the AAAI Conference on Artificial Intelligence.","volume":"36","author":"Jinze","year":"2022","unstructured":"Jinze Chen et al. ''Progressivemotionseg: Mutually reinforced framework for event-based motion segmentation''. In: Proceedings of the AAAI Conference on Artificial Intelligence. Vol. 36. 1. 2022, pp. 303--311."},{"key":"e_1_3_2_1_4_1","first-page":"3272","volume-title":"Proceedings of the 30th ACM International Conference on Multimedia.","author":"Zhi-Qi","year":"2022","unstructured":"Zhi-Qi Cheng et al. ''Gsrformer: Grounded situation recognition transformer with alternate semantic attention refinement''. In: Proceedings of the 30th ACM International Conference on Multimedia. 2022, pp. 3272--3281."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_5_1","DOI":"10.1109\/TCSVT.2017.2764624"},{"key":"e_1_3_2_1_6_1","first-page":"19659","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition.","author":"Cho Junhyeong","year":"2022","unstructured":"Junhyeong Cho, Youngseok Yoon, and Suha Kwak. ''Collaborative transformers for grounded situation recognition''. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 2022, pp. 19659--19668."},{"key":"e_1_3_2_1_7_1","volume-title":"Event extraction by answering (almost) natural questions''. In: arXiv preprint arXiv:2004.13625","author":"Du Xinya","year":"2020","unstructured":"Xinya Du and Claire Cardie. ''Event extraction by answering (almost) natural questions''. In: arXiv preprint arXiv:2004.13625 (2020)."},{"key":"e_1_3_2_1_8_1","first-page":"5504","volume-title":"Proceedings of the 31st ACM International Conference on Multimedia.","author":"Zilin","year":"2023","unstructured":"Zilin Du et al. ''Training multimedia event extraction with generated images and captions''. In: Proceedings of the 31st ACM International Conference on Multimedia. 2023, pp. 5504--5513."},{"key":"e_1_3_2_1_9_1","first-page":"5319","article-title":"DBiased-P: Dual-biased predicate predictor for unbiased scene graph generation","volume":"25","author":"Xianjing Han","year":"2022","unstructured":"Xianjing Han et al. ''DBiased-P: Dual-biased predicate predictor for unbiased scene graph generation''. In: IEEE Transactions on Multimedia 25 (2022), pp. 5319--5329.","journal-title":"IEEE Transactions on Multimedia"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_10_1","DOI":"10.1109\/TCSVT.2022.3193857"},{"key":"e_1_3_2_1_11_1","volume-title":"DEGREE: A data-efficient generation-based event extraction model''. In: arXiv preprint arXiv:2108.12724","author":"I Hsu","year":"2021","unstructured":"I Hsu et al. ''DEGREE: A data-efficient generation-based event extraction model''. In: arXiv preprint arXiv:2108.12724 (2021)."},{"key":"e_1_3_2_1_12_1","first-page":"254","volume-title":"Proceedings of ACL-08: Hlt.","author":"Ji Heng","year":"2008","unstructured":"Heng Ji and Ralph Grishman. ''Refining event extraction through cross-document inference''. In: Proceedings of ACL-08: Hlt. 2008, pp. 254--262."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_13_1","DOI":"10.1109\/JSTARS.2023.3289293"},{"key":"e_1_3_2_1_14_1","volume-title":"Deciding equivalances among conjunctive aggregate queries''. In: arXiv preprint arXiv:2210.03419","author":"Lai Viet Dac","year":"2022","unstructured":"Viet Dac Lai. ''Deciding equivalances among conjunctive aggregate queries''. In: arXiv preprint arXiv:2210.03419 (2022)."},{"key":"e_1_3_2_1_15_1","first-page":"829","volume-title":"Findings of the Association for Computational Linguistics: EMNLP 2020","author":"Fayuan","year":"2020","unstructured":"Fayuan Li et al. ''Event extraction as multi-turn question answering''. In: Findings of the Association for Computational Linguistics: EMNLP 2020. 2020, pp. 829--838."},{"key":"e_1_3_2_1_16_1","first-page":"16420","volume-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition.","author":"Manling","year":"2022","unstructured":"Manling Li et al. ''Clip-event: Connecting text and images with event structures''. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 2022, pp. 16420--16429."},{"key":"e_1_3_2_1_17_1","volume-title":"Cross-media structured common space for multimedia event extraction''. In: arXiv preprint arXiv:2005.02472","author":"Manling Li","year":"2020","unstructured":"Manling Li et al. ''Cross-media structured common space for multimedia event extraction''. In: arXiv preprint arXiv:2005.02472 (2020)."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_18_1","DOI":"10.1016\/j.neucom.2023.126433"},{"issue":"7","key":"e_1_3_2_1_19_1","first-page":"2155","article-title":"discriminative dictionary learning for image classification","volume":"30","author":"Ling Jing","year":"2019","unstructured":"Jing Ling, Zhenzhong Chen, and Feng Wu. ''Class-oriented discriminative dictionary learning for image classification''. In: IEEE Transactions on Circuits and Systems for Video Technology 30.7 (2019), pp. 2155--2166.","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_20_1","DOI":"10.1145\/3503161.3548132"},{"key":"e_1_3_2_1_21_1","volume-title":"Jointly multiple events extraction via attention-based graph information aggregation''. In: arXiv preprint arXiv:1809.09078","author":"Liu Xiao","year":"2018","unstructured":"Xiao Liu, Zhunchen Luo, and Heyan Huang. ''Jointly multiple events extraction via attention-based graph information aggregation''. In: arXiv preprint arXiv:1809.09078 (2018)."},{"key":"e_1_3_2_1_22_1","article-title":"Multi-grained gradual inference model for multimedia event extraction","author":"Yang Liu","year":"2024","unstructured":"Yang Liu et al. ''Multi-grained gradual inference model for multimedia event extraction''. In: IEEE Transactions on Circuits and Systems for Video Technology (2024).","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology ("},{"key":"e_1_3_2_1_23_1","volume-title":"Structured prediction as translation between augmented natural languages''. In: arXiv preprint arXiv:2101.05779","author":"Giovanni Paolini","year":"2021","unstructured":"Giovanni Paolini et al. ''Structured prediction as translation between augmented natural languages''. In: arXiv preprint arXiv:2101.05779 (2021)."},{"doi-asserted-by":"publisher","key":"e_1_3_2_1_24_1","DOI":"10.1109\/TCSVT.2020.3016863"},{"key":"e_1_3_2_1_25_1","first-page":"314","volume-title":"Proceedings, Part IV 16","author":"Sarah","year":"2020","unstructured":"Sarah Pratt et al. ''Grounded situation recognition''. In: Computer Vision--ECCV 2020: 16th European Conference, Glasgow, UK, August 23--28, 2020, Proceedings, Part IV 16. Springer. 2020, pp. 314--332."},{"issue":"2","key":"e_1_3_2_1_26_1","first-page":"549","article-title":"StagNet: An attentive semantic RNN for group activity and individual action recognition","volume":"30","author":"Mengshi Qi","year":"2019","unstructured":"Mengshi Qi et al. ''StagNet: An attentive semantic RNN for group activity and individual action recognition''. In: IEEE Transactions on Circuits and Systems for Video Technology 30.2 (2019), pp. 549--565.","journal-title":"IEEE Transactions on Circuits and Systems for Video Technology"},{"key":"e_1_3_2_1_27_1","first-page":"8748","volume-title":"International conference on machine learning. PmLR.","author":"Alec","year":"2021","unstructured":"Alec Radford et al. ''Learning transferable visual models from natural language supervision''. In: International conference on machine learning. PmLR. 2021, pp. 8748--8763."},{"key":"e_1_3_2_1_28_1","first-page":"19062","volume-title":"Proceedings of the AAAI Conference on Artificial Intelligence.","volume":"38","author":"Lin","year":"2024","unstructured":"Lin Sun et al. ''Umie: Unified multimodal information extraction with instruction tuning''. In: Proceedings of the AAAI Conference on Artificial Intelligence. Vol. 38. 17. 2024, pp. 19062--19070."},{"key":"e_1_3_2_1_29_1","first-page":"3716","volume-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition.","author":"Kaihua","year":"2020","unstructured":"Kaihua Tang et al. ''Unbiased scene graph generation from biased training''. In: Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 2020, pp. 3716--3725."},{"key":"e_1_3_2_1_30_1","volume-title":"Ace 2005 multilingual training corpus","author":"Christopher Walker","year":"2006","unstructured":"Christopher Walker et al. ''Ace 2005 multilingual training corpus''. In: (No Title) (2006)."},{"key":"e_1_3_2_1_31_1","volume-title":"Joint extraction of events and entities within a document context''. In: arXiv preprint arXiv:1609.03632","author":"Yang Bishan","year":"2016","unstructured":"Bishan Yang and Tom Mitchell. ''Joint extraction of events and entities within a document context''. In: arXiv preprint arXiv:1609.03632 (2016)."},{"key":"e_1_3_2_1_32_1","first-page":"5534","volume-title":"Proceedings of the IEEE conference on computer vision and pattern recognition.","author":"Yatskar Mark","year":"2016","unstructured":"Mark Yatskar, Luke Zettlemoyer, and Ali Farhadi. ''Situation recognition: Visual semantic role labeling for image understanding''. In: Proceedings of the IEEE conference on computer vision and pattern recognition. 2016, pp. 5534--5542."},{"key":"e_1_3_2_1_33_1","volume-title":"Event extraction with generative adversarial imitation learning''. In: arXiv preprint arXiv:1804.07881","author":"Zhang Tongtao","year":"2018","unstructured":"Tongtao Zhang and Heng Ji. ''Event extraction with generative adversarial imitation learning''. In: arXiv preprint arXiv:1804.07881 (2018)."}],"event":{"sponsor":["SIGIR ACM Special Interest Group on Information Retrieval","SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web"],"acronym":"CIKM '25","name":"CIKM '25: The 34th ACM International Conference on Information and Knowledge Management","location":"Seoul Republic of Korea"},"container-title":["Proceedings of the 34th ACM International Conference on Information and Knowledge Management"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746252.3761235","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,12]],"date-time":"2025-12-12T00:03:29Z","timestamp":1765497809000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746252.3761235"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,10]]},"references-count":33,"alternative-id":["10.1145\/3746252.3761235","10.1145\/3746252"],"URL":"https:\/\/doi.org\/10.1145\/3746252.3761235","relation":{},"subject":[],"published":{"date-parts":[[2025,11,10]]},"assertion":[{"value":"2025-11-10","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}