{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,5]],"date-time":"2026-06-05T15:24:11Z","timestamp":1780673051059,"version":"3.54.1"},"publisher-location":"Cham","reference-count":53,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031729324","type":"print"},{"value":"9783031729331","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,10,3]],"date-time":"2024-10-03T00:00:00Z","timestamp":1727913600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,3]],"date-time":"2024-10-03T00:00:00Z","timestamp":1727913600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-72933-1_23","type":"book-chapter","created":{"date-parts":[[2024,10,2]],"date-time":"2024-10-02T12:02:53Z","timestamp":1727870573000},"page":"402-418","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":6,"title":["Fast Encoding and\u00a0Decoding for\u00a0Implicit Video Representation"],"prefix":"10.1007","author":[{"given":"Hao","family":"Chen","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Saining","family":"Xie","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ser-Nam","family":"Lim","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Abhinav","family":"Shrivastava","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,10,3]]},"reference":[{"key":"23_CR1","doi-asserted-by":"crossref","unstructured":"Agustsson, E., Minnen, D., Johnston, N., Balle, J., Hwang, S.J., Toderici, G.: Scale-space flow for end-to-end optimized video compression. In: CVPR (2020)","DOI":"10.1109\/CVPR42600.2020.00853"},{"key":"23_CR2","doi-asserted-by":"crossref","unstructured":"Anokhin, I., Demochkin, K., Khakhulin, T., Sterkin, G., Lempitsky, V., Korzhenkov, D.: Image generators with conditionally-independent pixel synthesis. In: CVPR, pp. 14278\u201314287 (2021)","DOI":"10.1109\/CVPR46437.2021.01405"},{"issue":"10","key":"23_CR3","doi-asserted-by":"publisher","first-page":"3736","DOI":"10.1109\/TCSVT.2021.3101953","volume":"31","author":"B Bross","year":"2021","unstructured":"Bross, B., et al.: Overview of the versatile video coding (vvc) standard and its applications. IEEE Trans. Circuits Syst. Video Technol. 31(10), 3736\u20133764 (2021)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"23_CR4","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"213","DOI":"10.1007\/978-3-030-58452-8_13","volume-title":"Computer Vision \u2013 ECCV 2020","author":"N Carion","year":"2020","unstructured":"Carion, N., Massa, F., Synnaeve, G., Usunier, N., Kirillov, A., Zagoruyko, S.: End-to-End object detection with transformers. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12346, pp. 213\u2013229. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58452-8_13"},{"key":"23_CR5","doi-asserted-by":"crossref","unstructured":"Chen, H., Gwilliam, M., Lim, S.N., Shrivastava, A.: HNeRV: neural representations for videos. In: CVPR (2023)","DOI":"10.1109\/CVPR52729.2023.00990"},{"key":"23_CR6","unstructured":"Chen, H., He, B., Wang, H., Ren, Y., Lim, S.N., Shrivastava, A.: NeRV: neural representations for videos. In: NeurIPS (2021)"},{"key":"23_CR7","unstructured":"Chen, H., Matthew, A.G., He, B., Lim, S.N., Shrivastava, A.: CNeRV: content-adaptive neural representation for visual data. In: BMVC (2022)"},{"key":"23_CR8","doi-asserted-by":"crossref","unstructured":"Chen, Y., Liu, S., Wang, X.: Learning continuous image representation with local implicit image function. In: CVPR, pp. 8628\u20138638 (2021)","DOI":"10.1109\/CVPR46437.2021.00852"},{"key":"23_CR9","doi-asserted-by":"publisher","unstructured":"Chen, Y., Wang, X.: Transformers as meta-learners for implicit neural representations. In: European Conference on Computer Vision, pp. 170\u2013187. Springer, Heidelberg (2022). https:\/\/doi.org\/10.1007\/978-3-031-19790-1_11","DOI":"10.1007\/978-3-031-19790-1_11"},{"key":"23_CR10","doi-asserted-by":"crossref","unstructured":"Chen, Y., Dai, X., Liu, M., Chen, D., Yuan, L., Liu, Z.: Dynamic convolution: attention over convolution kernels. In: CVPR, pp. 11030\u201311039 (2020)","DOI":"10.1109\/CVPR42600.2020.01104"},{"key":"23_CR11","doi-asserted-by":"crossref","unstructured":"Chen, Y., et\u00a0al.: An overview of core coding tools in the av1 video codec. In: 2018 Picture Coding Symposium (PCS), pp. 41\u201345. IEEE (2018)","DOI":"10.1109\/PCS.2018.8456249"},{"key":"23_CR12","doi-asserted-by":"crossref","unstructured":"Chen, Z., et al.: Videoinr: learning video implicit neural representation for continuous space-time super-resolution. In: CVPR (2022)","DOI":"10.1109\/CVPR52688.2022.00209"},{"key":"23_CR13","doi-asserted-by":"crossref","unstructured":"Chen, Z., Zhang, H.: Learning implicit fields for generative shape modeling. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00609"},{"key":"23_CR14","doi-asserted-by":"crossref","unstructured":"Djelouah, A., Campos, J., Schaub-Meyer, S., Schroers, C.: Neural inter-frame compression for video coding. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00652"},{"key":"23_CR15","unstructured":"Dosovitskiy, A., et al.: An image is worth 16$$\\times $$16 words: transformers for image recognition at scale. In: ICLR (2021). https:\/\/openreview.net\/forum?id=YicbFdNTTy"},{"key":"23_CR16","unstructured":"Dupont, E., Goli\u0144ski, A., Alizadeh, M., Teh, Y.W., Doucet, A.: Coin: compression with implicit neural representations. In: ICLR workshop (2021)"},{"key":"23_CR17","first-page":"35946","volume":"35","author":"C Feichtenhofer","year":"2022","unstructured":"Feichtenhofer, C., Li, Y., He, K., et al.: Masked autoencoders as spatiotemporal learners. Adv. Neural. Inf. Process. Syst. 35, 35946\u201335958 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"23_CR18","doi-asserted-by":"crossref","unstructured":"Goyal, R., et al.: The \u201csomething something\" video database for learning and evaluating visual common sense. In: ICCV (2017)","DOI":"10.1109\/ICCV.2017.622"},{"key":"23_CR19","unstructured":"Ha, D., Dai, A.M., Le, Q.V.: Hypernetworks. In: ICLR (2017)"},{"key":"23_CR20","doi-asserted-by":"crossref","unstructured":"Habibian, A., Rozendaal, T.v., Tomczak, J.M., Cohen, T.S.: Video compression with rate-distortion autoencoders. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00713"},{"key":"23_CR21","doi-asserted-by":"crossref","unstructured":"He, B., et al.: Towards scalable neural representation for diverse videos. In: CVPR (2023)","DOI":"10.1109\/CVPR52729.2023.00594"},{"key":"23_CR22","doi-asserted-by":"crossref","unstructured":"He, K., Chen, X., Xie, S., Li, Y., Doll\u00e1r, P., Girshick, R.: Masked autoencoders are scalable vision learners. In: CVPR (2022)","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"23_CR23","unstructured":"Karras, T., et al.: Alias-free generative adversarial networks. arXiv preprint arXiv:2106.12423 (2021)"},{"key":"23_CR24","unstructured":"Kay, W., et\u00a0al.: The kinetics human action video dataset. arXiv preprint arXiv:1705.06950 (2017)"},{"key":"23_CR25","doi-asserted-by":"crossref","unstructured":"Khani, M., Sivaraman, V., Alizadeh, M.: Efficient video compression via content-adaptive super-resolution. arXiv preprint arXiv:2104.02322 (2021)","DOI":"10.1109\/ICCV48922.2021.00448"},{"key":"23_CR26","doi-asserted-by":"crossref","unstructured":"Kim, C., Lee, D., Kim, S., Cho, M., Han, W.S.: Generalizable implicit neural representations via instance pattern composers. In: CVPR (2023)","DOI":"10.1109\/CVPR52729.2023.01136"},{"key":"23_CR27","unstructured":"Kim, S., Yu, S., Lee, J., Shin, J.: Scalable neural video representations with learnable positional features. In: NeurIPS (2022)"},{"key":"23_CR28","doi-asserted-by":"publisher","first-page":"46","DOI":"10.1145\/103085.103090","volume":"34","author":"D Le Gall","year":"1991","unstructured":"Le Gall, D.: MPEG: a video compression standard for multimedia applications. ACM Commun. 34, 46\u201358 (1991)","journal-title":"ACM Commun."},{"key":"23_CR29","doi-asserted-by":"publisher","first-page":"267","DOI":"10.1007\/978-3-031-19833-5_16","volume-title":"ECCV 2022","author":"Z Li","year":"2022","unstructured":"Li, Z., Wang, M., Pi, H., Xu, K., Mei, J., Liu, Y.: E-nerv: expedite neural video representation with disentangled spatial-temporal context. In: Avidan, S., Brostow, G., Cisse, M., Farinella, G.M., Hassner, T. (eds.) ECCV 2022, vol. 13695, pp. 267\u2013284. Springer, Heidelberg (2022). https:\/\/doi.org\/10.1007\/978-3-031-19833-5_16"},{"key":"23_CR30","unstructured":"Liu, H., Chen, T., Lu, M., Shen, Q., Ma, Z.: Neural video compression using spatio-temporal priors. arXiv preprint arXiv:1902.07383 (2019)"},{"key":"23_CR31","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"453","DOI":"10.1007\/978-3-030-58520-4_27","volume-title":"Computer Vision \u2013 ECCV 2020","author":"J Liu","year":"2020","unstructured":"Liu, J., et al.: Conditional entropy coding for efficient video compression. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12362, pp. 453\u2013468. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58520-4_27"},{"key":"23_CR32","doi-asserted-by":"crossref","unstructured":"Liu, Z., et al.: Swin transformer: hierarchical vision transformer using shifted windows. In: ICCV (2021)","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"23_CR33","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101 (2017)"},{"key":"23_CR34","doi-asserted-by":"crossref","unstructured":"Mehta, I., Gharbi, M., Barnes, C., Shechtman, E., Ramamoorthi, R., Chandraker, M.: Modulated periodic activations for generalizable local functional representations. In: ICCV, pp. 14214\u201314223 (2021)","DOI":"10.1109\/ICCV48922.2021.01395"},{"key":"23_CR35","doi-asserted-by":"crossref","unstructured":"Mescheder, L., Oechsle, M., Niemeyer, M., Nowozin, S., Geiger, A.: Occupancy networks: learning 3d reconstruction in function space. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00459"},{"key":"23_CR36","series-title":"Lecture Notes in Computer Science","doi-asserted-by":"publisher","first-page":"405","DOI":"10.1007\/978-3-030-58452-8_24","volume-title":"Computer Vision \u2013 ECCV 2020","author":"B Mildenhall","year":"2020","unstructured":"Mildenhall, B., Srinivasan, P.P., Tancik, M., Barron, J.T., Ramamoorthi, R., Ng, R.: NeRF: representing scenes as neural radiance fields for view synthesis. In: Vedaldi, A., Bischof, H., Brox, T., Frahm, J.-M. (eds.) ECCV 2020. LNCS, vol. 12346, pp. 405\u2013421. Springer, Cham (2020). https:\/\/doi.org\/10.1007\/978-3-030-58452-8_24"},{"key":"23_CR37","doi-asserted-by":"crossref","unstructured":"Park, J.J., Florence, P., Straub, J., Newcombe, R., Lovegrove, S.: Deepsdf: learning continuous signed distance functions for shape representation. In: CVPR (2019)","DOI":"10.1109\/CVPR.2019.00025"},{"key":"23_CR38","unstructured":"Paszke, A., et al.: Pytorch: an imperative style, high-performance deep learning library. In: NeurIPS (2019)"},{"key":"23_CR39","doi-asserted-by":"crossref","unstructured":"Rippel, O., Anderson, A.G., Tatwawadi, K., Nair, S., Lytle, C., Bourdev, L.: Elf-vc: efficient learned flexible-rate video coding. In: ICCV (2021)","DOI":"10.1109\/ICCV48922.2021.01421"},{"key":"23_CR40","doi-asserted-by":"crossref","unstructured":"Rippel, O., Nair, S., Lew, C., Branson, S., Anderson, A.G., Bourdev, L.: Learned video compression. In: ICCV (2019)","DOI":"10.1109\/ICCV.2019.00355"},{"key":"23_CR41","unstructured":"Schwarz, K., Liao, Y., Niemeyer, M., Geiger, A.: Graf: generative radiance fields for 3d-aware image synthesis. In: NeurIPS (2020)"},{"key":"23_CR42","unstructured":"Sitzmann, V., Martel, J.N.P., Bergman, A.W., Lindell, D.B., Wetzstein, G.: Implicit neural representations with periodic activation functions. In: NeurIPS (2020)"},{"key":"23_CR43","unstructured":"Sitzmann, V., Martel, J.N., Bergman, A.W., Lindell, D.B., Wetzstein, G.: Implicit neural representations with periodic activation functions. In: NeurIPS (2020)"},{"key":"23_CR44","unstructured":"Sitzmann, V., Zollh\u00f6fer, M., Wetzstein, G.: Scene representation networks: continuous 3d-structure-aware neural scene representations. In: NeurIPS (2019)"},{"key":"23_CR45","doi-asserted-by":"crossref","unstructured":"Skorokhodov, I., Ignatyev, S., Elhoseiny, M.: Adversarial generation of continuous images. In: CVPR, pp. 10753\u201310764 (2021)","DOI":"10.1109\/CVPR46437.2021.01061"},{"key":"23_CR46","unstructured":"Soomro, K., Za4mir, A.R., Shah, M.: Ucf101: a dataset of 101 human actions classes from videos in the wild. arXiv preprint arXiv:1212.0402 (2012)"},{"key":"23_CR47","doi-asserted-by":"publisher","first-page":"1649","DOI":"10.1109\/TCSVT.2012.2221191","volume":"22","author":"GJ Sullivan","year":"2012","unstructured":"Sullivan, G.J., Ohm, J.R., Han, W.J., Wiegand, T.: Overview of the high efficiency video coding (hevc) standard. IEEE Trans. Circuits Syst. Video Technol. 22, 1649\u20131668 (2012)","journal-title":"IEEE Trans. Circuits Syst. Video Technol."},{"key":"23_CR48","unstructured":"Tancik, M., et al.: Fourier features let networks learn high frequency functions in low dimensional domains. In: NeurIPS (2020)"},{"key":"23_CR49","unstructured":"Touvron, H., Cord, M., Douze, M., Massa, F., Sablayrolles, A., J\u00e9gou, H.: Training data-efficient image transformers & distillation through attention. In: ICML, pp. 10347\u201310357. PMLR (2021)"},{"key":"23_CR50","doi-asserted-by":"crossref","unstructured":"Wiegand, T., Sullivan, G.J., Bjontegaard, G., Luthra, A.: Overview of the h. 264\/avc video coding standard. IEEE Trans. Circuits Syst. Video Technol. 13, 560\u2013576 (2003)","DOI":"10.1109\/TCSVT.2003.815165"},{"key":"23_CR51","doi-asserted-by":"crossref","unstructured":"Wu, C.Y., Girshick, R., He, K., Feichtenhofer, C., Krahenbuhl, P.: A multigrid method for efficiently training video models. In: CVPR, pp. 153\u2013162 (2020)","DOI":"10.1109\/CVPR42600.2020.00023"},{"key":"23_CR52","doi-asserted-by":"crossref","unstructured":"Wu, C.Y., Singhal, N., Krahenbuhl, P.: Video compression through image interpolation. In: ECCV (2018)","DOI":"10.1007\/978-3-030-01237-3_26"},{"key":"23_CR53","unstructured":"Yang, B., Bender, G., Le, Q.V., Ngiam, J.: Condconv: conditionally parameterized convolutions for efficient inference. In: NeurIPS, vol. 32 (2019)"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-72933-1_23","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,2]],"date-time":"2024-10-02T12:38:15Z","timestamp":1727872695000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-72933-1_23"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,3]]},"ISBN":["9783031729324","9783031729331"],"references-count":53,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-72933-1_23","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,3]]},"assertion":[{"value":"3 October 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}