{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,18]],"date-time":"2026-06-18T08:51:15Z","timestamp":1781772675851,"version":"3.54.5"},"reference-count":24,"publisher":"China Science Publishing & Media Ltd.","issue":"2","content-domain":{"domain":["engine.scichina.com"],"crossmark-restriction":false},"short-container-title":["DI"],"published-print":{"date-parts":[[2026,6,1]]},"DOI":"10.3724\/2096-7004.di.2025.0120","type":"journal-article","created":{"date-parts":[[2026,6,8]],"date-time":"2026-06-08T08:11:18Z","timestamp":1780906278000},"page":"20250120","update-policy":"https:\/\/doi.org\/10.1360\/scp-crossmark-policy-page","source":"Crossref","is-referenced-by-count":0,"title":["TCAR-Net: Text-Driven Compressed Action Recognition with Multimodal Fusion"],"prefix":"10.3724","volume":"8","author":[{"given":"Mengkun","family":"Guo","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xinqi","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Die","family":"Tao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ming","family":"Ma","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"2026","published-online":{"date-parts":[[2026,6,8]]},"reference":[{"key":"null","unstructured":"Poquet O., Lim L., Mirriahi N., and Dawson S., \u201cVideo and learning: A systematic review (2007\u20132017),\u201d in Proc. 8th Int. Conf. Learn. Analytics Knowl., 2018, pp. 151\u2013160."},{"key":"null","unstructured":"Simonyan K. and Zisserman A., \u201cTwo-stream convolutional networks for action recognition in videos,\u201d in Adv. Neural Inf. Process. Syst., vol. 27, 2014."},{"key":"null","unstructured":"Carreira J. and Zisserman A., \u201cQuo vadis, action recognition? A new model and the kinetics dataset,\u201d in Proc. IEEE Conf. Comput. Vis. Pattern Recognit. (CVPR), 2017, pp. 6299\u20136308."},{"key":"null","unstructured":"Tran D., Wang H., Torresani L., Ray J., LeCun Y., and Paluri M., \u201cA closer look at spatiotemporal convolutions for action recognition,\u201d in Proc. IEEE Conf. Comput. Vis. Pattern Recognit. (CVPR), 2018, pp. 6450\u20136459."},{"key":"null","unstructured":"Wu C.-Y., Zaheer M., Hu H., Manmatha R., Smola A. J., and Krahenbuhl P., \u201cCompressed video action recognition,\u201d in Proc. IEEE Conf. Comput. Vis. Pattern Recognit. (CVPR), 2018, pp. 6026\u20136035."},{"key":"null","unstructured":"Shou Z. et al., \u201cDMC-Net: Generating discriminative motion cues for fast compressed video action recognition,\u201d in Proc. IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. (CVPR), 2019, pp. 1268\u20131277."},{"key":"null","unstructured":"Huang S., Lin X., Karaman S., and Chang S.-F., \u201cFlow-distilled IP two-stream networks for compressed video action recognition,\u201d arXiv preprint, arXiv:1912.04462, 2019."},{"key":"null","unstructured":"Battash B., Barad H., Tang H., and Bleiweiss A., \u201cMimic the raw domain: Accelerating action recognition in the compressed domain,\u201d in Proc. IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. Workshops (CVPRW), 2020, pp. 684\u2013685."},{"key":"null","unstructured":"Cao H., Yu S., and Feng J., \u201cCompressed video action recognition with refined motion vector,\u201d arXiv preprint, arXiv:1910.02533, 2019."},{"key":"null","unstructured":"Radford A. et al., \u201cLearning transferable visual models from natural language supervision,\u201d in Proc. Int. Conf. Mach. Learn. (ICML), 2021, pp. 8748\u20138763."},{"key":"null","unstructured":"Wu W., Wang X., Luo H., Wang J., Yang Y., and Ouyang W., \u201cBidirectional cross-modal knowledge exploration for video recognition with pre-trained vision-language models,\u201d in Proc. IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. (CVPR), 2023, pp. 6620\u20136630."},{"key":"null","unstructured":"Huo Y., Xu X., Lu Y., Niu Y., Lu Z., and Wen J.-R., \u201cMobile video action recognition,\u201d arXiv preprint, arXiv:1908.10155, 2019."},{"key":"null","unstructured":"Feichtenhofer C., Fan H., Malik J., and He K., \u201cSlowFast networks for video recognition,\u201d in Proc. IEEE\/CVF Int. Conf. Comput. Vis. (ICCV), 2019, pp. 6202\u20136211."},{"key":"null","unstructured":"Li J., Wei P., Zhang Y., and Zheng N., \u201cA slow-I-fast-P architecture for compressed video action recognition,\u201d in Proc. 28th ACM Int. Conf. Multimedia, 2020, pp. 2039\u20132047."},{"key":"null","unstructured":"Chen J. and Ho C. M., \u201cMM-ViT: Multi-modal video transformer for compressed video action recognition,\u201d in Proc. IEEE\/CVF Winter Conf. Appl. Comput. Vis. (WACV), 2022, pp. 1910\u20131921."},{"key":"null","unstructured":"Yuan L. et al., \u201cFlorence: A new foundation model for computer vision,\u201d arXiv preprint, arXiv:2111.11432, 2021."},{"key":"null","unstructured":"Wang M., Xing J., and Liu Y., \u201cActionCLIP: A new paradigm for video action recognition,\u201d arXiv preprint, arXiv:2109.08472, 2021."},{"key":"null","unstructured":"Ju C., Han T., Zheng K., Zhang Y., and Xie W., \u201cPrompting visual-language models for efficient video understanding,\u201d in Proc. Eur. Conf. Comput. Vis. (ECCV), 2022, pp. 105\u2013124."},{"key":"null","unstructured":"Wang L., Tong Z., Ji B., and Wu G., \u201cTDN: Temporal difference networks for efficient action recognition,\u201d in Proc. IEEE\/CVF Conf. Comput. Vis. Pattern Recognit. (CVPR), 2021, pp. 1895\u20131904."},{"key":"null","unstructured":"Wang L., Li W., Li W., and Van Gool L., \u201cAppearance-and-relation networks for video classification,\u201d in Proc. IEEE Conf. Comput. Vis. Pattern Recognit. (CVPR), 2018, pp. 1430\u20131439."},{"key":"null","unstructured":"Tong Z., Song Y., Wang J., and Wang L., \u201cVideoMAE: Masked autoencoders are data-efficient learners for self-supervised video pre-training,\u201d in Adv. Neural Inf. Process. Syst., vol. 35, 2022, pp. 10078\u201310093."},{"key":"null","unstructured":"Wu W., He D., Lin T., Li F., Gan C., and Ding E., \u201cMVFNet: Multi-view fusion network for efficient action recognition,\u201d in Proc. AAAI Conf. Artif. Intell., 2021."},{"key":"null","unstructured":"Wu W., Sun Z., and Ouyang W., \u201cRevisiting classifier: Transferring vision-language models for video recognition,\u201d in Proc. AAAI Conf. Artif. Intell., vol. 37, 2023, pp. 2847\u20132855."},{"key":"null","unstructured":"Li B., Chen J., Bao X., and Huang D., \u201cCompressed video prompt tuning,\u201d in Adv. Neural Inf. Process. Syst., vol. 36, 2023, pp. 31895\u201331907."}],"container-title":["Data Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/www.sciengine.com\/sci-open\/api\/v1\/open\/file\/pdf\/089412AB1AC84D70A757B367EBDD8DB3","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/www.sciengine.com\/doi\/10.3724\/2096-7004.di.2025.0120","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/www.sciengine.com\/sci-open\/api\/v1\/open\/file\/pdf\/089412AB1AC84D70A757B367EBDD8DB3","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,6,18]],"date-time":"2026-06-18T07:56:36Z","timestamp":1781769396000},"score":1,"resource":{"primary":{"URL":"https:\/\/www.sciengine.com\/doi\/10.3724\/2096-7004.di.2025.0120"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6,1]]},"references-count":24,"journal-issue":{"issue":"2","published-online":{"date-parts":[[2026,6,8]]},"published-print":{"date-parts":[[2026,6,1]]}},"URL":"https:\/\/doi.org\/10.3724\/2096-7004.di.2025.0120","relation":{},"ISSN":["2096-7004"],"issn-type":[{"value":"2096-7004","type":"print"}],"subject":[],"published":{"date-parts":[[2026,6,1]]}}}