{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,22]],"date-time":"2026-04-22T19:12:40Z","timestamp":1776885160990,"version":"3.51.2"},"publisher-location":"New York, NY, USA","reference-count":42,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,10,26]],"date-time":"2023-10-26T00:00:00Z","timestamp":1698278400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"China University Innovation Fund","award":["2021FNA04003"],"award-info":[{"award-number":["2021FNA04003"]}]},{"name":"Project of China Knowledge Centre for Engineering Science and Technology"},{"name":"NSFC under Grant","award":["62172326, 62137002"],"award-info":[{"award-number":["62172326, 62137002"]}]},{"name":"the MOE Innovation Research Team","award":["IRT17R86"],"award-info":[{"award-number":["IRT17R86"]}]},{"name":"National Key R&D Program of China","award":["2020AAA0108800"],"award-info":[{"award-number":["2020AAA0108800"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,10,26]]},"DOI":"10.1145\/3581783.3613809","type":"proceedings-article","created":{"date-parts":[[2023,10,27]],"date-time":"2023-10-27T07:27:12Z","timestamp":1698391632000},"page":"3560-3568","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":14,"title":["Tile Classification Based Viewport Prediction with Multi-modal Fusion Transformer"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-3791-8785","authenticated-orcid":false,"given":"Zhiahao","family":"Zhang","sequence":"first","affiliation":[{"name":"Xi'an Jiaotong University, Xi'an, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0219-9143","authenticated-orcid":false,"given":"Yiwei","family":"Chen","sequence":"additional","affiliation":[{"name":"Xi'an Jiaotong University, Xi'an, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0330-5435","authenticated-orcid":false,"given":"Weizhan","family":"Zhang","sequence":"additional","affiliation":[{"name":"Xi'an Jiaotong University, Xi'an, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7763-2987","authenticated-orcid":false,"given":"Caixia","family":"Yan","sequence":"additional","affiliation":[{"name":"Xi'an Jiaotong University, Xi'an, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8436-4754","authenticated-orcid":false,"given":"Qinghua","family":"Zheng","sequence":"additional","affiliation":[{"name":"Xi'an Jiaotong University, Xi'an, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0639-2119","authenticated-orcid":false,"given":"Qi","family":"Wang","sequence":"additional","affiliation":[{"name":"Xi'an Jiaotong University, Xi'an, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8921-0092","authenticated-orcid":false,"given":"Wangdu","family":"Chen","sequence":"additional","affiliation":[{"name":"MIGU Video, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,10,27]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/IC3D.2017.8251913"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICME.2018.8486606"},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-58452-8_13"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"crossref","unstructured":"Fang-Yi Chao Cagri Ozcinar and Aljosa Smolic. 2021. Transformer-based Long-Term Viewport Prediction in 360\u00b0 Video: Scanpath is All You Need.. In MMSP. 1--6.","DOI":"10.1109\/MMSP53017.2021.9733647"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2020.3033127"},{"key":"e_1_3_2_2_6_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.comcom.2021.06.029"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3442381.3450070"},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/JETCAS.2019.2899516"},{"key":"e_1_3_2_2_9_1","volume-title":"Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805","author":"Devlin Jacob","year":"2018","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2018. Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805 (2018)."},{"key":"e_1_3_2_2_10_1","unstructured":"Alexey Dosovitskiy Lucas Beyer Alexander Kolesnikov Dirk Weissenborn Xiaohua Zhai Thomas Unterthiner Mostafa Dehghani Matthias Minderer Georg Heigold Sylvain Gelly et al. 2020. An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)."},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3328914"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"crossref","unstructured":"Kai Han Yunhe Wang Hanting Chen Xinghao Chen Jianyuan Guo Zhenhua Liu Yehui Tang An Xiao Chunjing Xu Yixing Xu et al. 2022. A survey on vision transformer. IEEE transactions on pattern analysis and machine intelligence Vol. 45 1 (2022) 87--110.","DOI":"10.1109\/TPAMI.2022.3152247"},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCAS45731.2020.9180528"},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2020.3022062"},{"key":"e_1_3_2_2_15_1","volume-title":"Robust and Resource-efficient Machine Learning Aided Viewport Prediction in Virtual Reality. arXiv preprint arXiv:2212.09945","author":"Jiang Yuang","year":"2022","unstructured":"Yuang Jiang, Konstantinos Poularakis, Diego Kiedanski, Sastry Kompella, and Leandros Tassiulas. 2022. Robust and Resource-efficient Machine Learning Aided Viewport Prediction in Virtual Reality. arXiv preprint arXiv:2212.09945 (2022)."},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1080\/23270012.2020.1756939"},{"key":"e_1_3_2_2_17_1","volume-title":"International Conference on Machine Learning. PMLR, 5583--5594","author":"Kim Wonjae","year":"2021","unstructured":"Wonjae Kim, Bokyung Son, and Ildoo Kim. 2021. Vilt: Vision-and-language transformer without convolution or region supervision. In International Conference on Machine Learning. PMLR, 5583--5594."},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.3390\/s21113678"},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/MIPR.2019.00060"},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i07.6795"},{"key":"e_1_3_2_2_21_1","volume-title":"Spherical Convolution empowered Viewport Prediction in 360 Video Multicast with Limited FoV Feedback. ACM Transactions on Multimedia Computing, Communications, and Applications (TOMM)","author":"Li Jie","year":"2022","unstructured":"Jie Li, Ling Han, Chong Zhang, Qiyue Li, and Zhi Liu. 2022. Spherical Convolution empowered Viewport Prediction in 360 Video Multicast with Limited FoV Feedback. ACM Transactions on Multimedia Computing, Communications, and Applications (TOMM) (2022)."},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00320"},{"key":"e_1_3_2_2_24_1","volume-title":"Vilbert: Pretraining task-agnostic visiolinguistic representations for vision-and-language tasks. Advances in neural information processing systems","author":"Lu Jiasen","year":"2019","unstructured":"Jiasen Lu, Dhruv Batra, Devi Parikh, and Stefan Lee. 2019. Vilbert: Pretraining task-agnostic visiolinguistic representations for vision-and-language tasks. Advances in neural information processing systems, Vol. 32 (2019)."},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01045"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1145\/3386290.3396934"},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/3240508.3240669"},{"key":"e_1_3_2_2_28_1","unstructured":"Colin Raffel Noam Shazeer Adam Roberts Katherine Lee Sharan Narang Michael Matena Yanqi Zhou Wei Li Peter J Liu et al. 2019. Exploring the limits of transfer learning with a unified text-to-text transformer. arXiv preprint arXiv:1910.10683 (2019)."},{"key":"e_1_3_2_2_29_1","volume-title":"Ramon Aparicio Pardo, and Frederic Precioso","author":"Romero Rondon Miguel Fabian","year":"2019","unstructured":"Miguel Fabian Romero Rondon, Lucile Sassatelli, Ramon Aparicio Pardo, and Frederic Precioso. 2019. Revisiting Deep Architectures for Head Motion Prediction in 360\u00b0 Videos. arXiv preprint arXiv:1911.11702 (2019)."},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00474"},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3474833"},{"key":"e_1_3_2_2_32_1","volume-title":"2019 IFIP\/IEEE Symposium on Integrated Network and Service Management (IM). IEEE, 381--387","author":"van der Hooft Jeroen","year":"2019","unstructured":"Jeroen van der Hooft, Maria Torres Vega, Stefano Petrangeli, Tim Wauters, and Filip De Turck. 2019. Optimizing adaptive tile-based virtual reality video streaming. In 2019 IFIP\/IEEE Symposium on Integrated Network and Service Management (IM). IEEE, 381--387."},{"key":"e_1_3_2_2_33_1","volume-title":"Advances in Neural Information Processing Systems","volume":"30","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. In Advances in Neural Information Processing Systems, Vol. 30."},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3123266.3123291"},{"key":"e_1_3_2_2_35_1","volume-title":"Predicting head movement in panoramic video: A deep reinforcement learning approach","author":"Xu Mai","year":"2018","unstructured":"Mai Xu, Yuhang Song, Jianyi Wang, MingLang Qiao, Liangyu Huo, and Zulin Wang. 2018b. Predicting head movement in panoramic video: A deep reinforcement learning approach. IEEE transactions on pattern analysis and machine intelligence, Vol. 41, 11 (2018), 2693--2708."},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/3359989.3365413"},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00559"},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCAS.2019.8702654"},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.1109\/COMST.2020.3006999"},{"key":"e_1_3_2_2_40_1","volume-title":"A combined field-of-view prediction-assisted viewport adaptive delivery scheme for 360\u00b0 videosIEEE Transactions on Broadcasting","author":"Yaqoob Abid","year":"2021","unstructured":"Abid Yaqoob and Gabriel-Miro Muntean. 2021. A combined field-of-view prediction-assisted viewport adaptive delivery scheme for 360\u00b0 videosIEEE Transactions on Broadcasting, Vol. 67, 3 (2021), 746--760."},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICME52920.2022.9859789"},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2019.2957986"}],"event":{"name":"MM '23: The 31st ACM International Conference on Multimedia","location":"Ottawa ON Canada","acronym":"MM '23","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 31st ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3613809","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3581783.3613809","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,21]],"date-time":"2025-08-21T23:59:16Z","timestamp":1755820756000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3581783.3613809"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,10,26]]},"references-count":42,"alternative-id":["10.1145\/3581783.3613809","10.1145\/3581783"],"URL":"https:\/\/doi.org\/10.1145\/3581783.3613809","relation":{},"subject":[],"published":{"date-parts":[[2023,10,26]]},"assertion":[{"value":"2023-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}