{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,4]],"date-time":"2026-06-04T13:20:00Z","timestamp":1780579200193,"version":"3.54.1"},"publisher-location":"Cham","reference-count":59,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783031729454","type":"print"},{"value":"9783031729461","type":"electronic"}],"license":[{"start":{"date-parts":[[2024,10,2]],"date-time":"2024-10-02T00:00:00Z","timestamp":1727827200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,2]],"date-time":"2024-10-02T00:00:00Z","timestamp":1727827200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-72946-1_27","type":"book-chapter","created":{"date-parts":[[2024,10,1]],"date-time":"2024-10-01T19:02:08Z","timestamp":1727809328000},"page":"478-495","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":13,"title":["Diffusion Reward: Learning Rewards via\u00a0Conditional Video Diffusion"],"prefix":"10.1007","author":[{"given":"Tao","family":"Huang","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Guangqi","family":"Jiang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yanjie","family":"Ze","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Huazhe","family":"Xu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,10,2]]},"reference":[{"key":"27_CR1","unstructured":"Adeniji, A., Xie, A., Sferrazza, C., Seo, Y., James, S., Abbeel, P.: Language reward modulation for pretraining reinforcement learning. arXiv preprint arXiv:2308.12270 (2023)"},{"key":"27_CR2","unstructured":"Ajay, A., Du, Y., Gupta, A., Tenenbaum, J., Jaakkola, T., Agrawal, P.: Is conditional generative modeling all you need for decision-making? In: International Conference on Learning Representations (ICLR) (2023)"},{"key":"27_CR3","doi-asserted-by":"crossref","unstructured":"Alakuijala, M., Dulac-Arnold, G., Mairal, J., Ponce, J., Schmid, C.: Learning reward functions for robotic manipulation by observing humans. In: IEEE International Conference on Robotics and Automation (ICRA) (2022)","DOI":"10.1109\/ICRA48891.2023.10161178"},{"key":"27_CR4","unstructured":"Burda, Y., Edwards, H., Storkey, A.J., Klimov, O.: Exploration by random network distillation. In: International Conference on Learning Representations (ICLR) (2019)"},{"key":"27_CR5","doi-asserted-by":"crossref","unstructured":"Ceylan, D., Huang, C.H.P., Mitra, N.J.: Pix2Video: video editing using image diffusion. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV) (2023)","DOI":"10.1109\/ICCV51070.2023.02121"},{"key":"27_CR6","doi-asserted-by":"crossref","unstructured":"Chen, A.S., Nair, S., Finn, C.: Learning generalizable robotic reward functions from \u201cin-the-wild\u201d human videos. In: Robotics: Science and Systems (RSS) (2021)","DOI":"10.15607\/RSS.2021.XVII.012"},{"key":"27_CR7","doi-asserted-by":"crossref","unstructured":"Chen, Z., Kiami, S., Gupta, A., Kumar, V.: GenAug: retargeting behaviors to unseen situations via generative augmentation. arXiv preprint arXiv:2302.06671 (2023)","DOI":"10.15607\/RSS.2023.XIX.010"},{"key":"27_CR8","doi-asserted-by":"crossref","unstructured":"Chi, C., et al.: Diffusion policy: visuomotor policy learning via action diffusion. In: Robotics: Science and Systems (RSS) (2023)","DOI":"10.15607\/RSS.2023.XIX.026"},{"key":"27_CR9","unstructured":"Du, Y., et al.: Learning universal policies via text-guided video generation. In: Advances in Neural Information Processing Systems (NeurIPS) (2023)"},{"key":"27_CR10","unstructured":"Escontrela, A., et al.: Video prediction models as rewards for reinforcement learning. In: Advances in Neural Information Processing Systems (NeurIPS) (2023)"},{"key":"27_CR11","doi-asserted-by":"crossref","unstructured":"Esser, P., Chiu, J., Atighehchian, P., Granskog, J., Germanidis, A.: Structure and content-guided video synthesis with diffusion models. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (ICCV) (2023)","DOI":"10.1109\/ICCV51070.2023.00675"},{"key":"27_CR12","doi-asserted-by":"crossref","unstructured":"Esser, P., Rombach, R., Ommer, B.: Taming transformers for high-resolution image synthesis. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2021)","DOI":"10.1109\/CVPR46437.2021.01268"},{"key":"27_CR13","doi-asserted-by":"crossref","unstructured":"Gu, S., et al.: Vector quantized diffusion model for text-to-image synthesis. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2022)","DOI":"10.1109\/CVPR52688.2022.01043"},{"key":"27_CR14","unstructured":"Hansen-Estruch, P., Kostrikov, I., Janner, M., Kuba, J.G., Levine, S.: IDQL: implicit Q-learning as an actor-critic method with diffusion policies. arXiv preprint arXiv:2304.10573 (2023)"},{"key":"27_CR15","unstructured":"Ho, J., et\u00a0al.: Imagen video: high definition video generation with diffusion models. arXiv preprint arXiv:2210.02303 (2022)"},{"key":"27_CR16","unstructured":"Ho, J., Jain, A., Abbeel, P.: Denoising diffusion probabilistic models. In: Advances in Neural Information Processing Systems (NeurIPS) (2020)"},{"key":"27_CR17","unstructured":"Ho, J., Salimans, T., Gritsenko, A., Chan, W., Norouzi, M., Fleet, D.J.: Video diffusion models. arXiv preprint arXiv:2204.03458 (2022)"},{"key":"27_CR18","unstructured":"Hu, J., et al.: Instructed diffuser with temporal condition guidance for offline reinforcement learning. arXiv preprint arXiv:2306.04875 (2023)"},{"key":"27_CR19","unstructured":"Janner, M., Du, Y., Tenenbaum, J.B., Levine, S.: Planning with diffusion for flexible behavior synthesis. In: International Conference on Machine Learning (ICML) (2022)"},{"key":"27_CR20","unstructured":"Kingma, D.P., Welling, M.: Auto-encoding variational bayes. arXiv preprint arXiv:1312.6114 (2013)"},{"key":"27_CR21","unstructured":"Ko, P.C., Mao, J., Du, Y., Sun, S.H., Tenenbaum, J.B.: Learning to act from actionless video through dense correspondences. arXiv preprint arXiv:2310.08576 (2023)"},{"key":"27_CR22","unstructured":"Kun, L., He, Z., Lu, C., Hu, K., Gao, Y., Xu, H.: Uni-O4: unifying online and offline deep reinforcement learning with multi-step on-policy optimization. In: International Conference on Learning Representations (ICLR) (2023)"},{"key":"27_CR23","unstructured":"Lee, S., Seo, Y., Lee, K., Abbeel, P., Shin, J.: Offline-to-online reinforcement learning via balanced replay and pessimistic Q-ensemble. In: Conference on Robot Learning (CoRL) (2022)"},{"key":"27_CR24","doi-asserted-by":"crossref","unstructured":"Lynch, C., Sermanet, P.: Language conditioned imitation learning over unstructured data. In: Robotics: Science and Systems (RSS) (2020)","DOI":"10.15607\/RSS.2021.XVII.047"},{"key":"27_CR25","unstructured":"Ma, Y.J., Sodhani, S., Jayaraman, D., Bastani, O., Kumar, V., Zhang, A.: VIP: towards universal visual reward and representation via value-implicit pre-training. In: International Conference on Learning Representations (ICLR) (2023)"},{"key":"27_CR26","unstructured":"Molad, E., et al.: Dreamix: video diffusion models are general video editors. arXiv preprint arXiv:2302.01329 (2023)"},{"key":"27_CR27","unstructured":"Nair, A., Gupta, A., Dalal, M., Levine, S.: AWAC: accelerating online reinforcement learning with offline datasets. arXiv preprint arXiv:2006.09359 (2020)"},{"key":"27_CR28","unstructured":"Nair, S., Rajeswaran, A., Kumar, V., Finn, C., Gupta, A.: R3M: a universal visual representation for robot manipulation. In: Conference on Robot Learning (CoRL) (2023)"},{"key":"27_CR29","unstructured":"Nasiriany, S., Pong, V., Lin, S., Levine, S.: Planning with goal-conditioned policies. In: Advances in Neural Information Processing Systems (NeurIPS) (2019)"},{"key":"27_CR30","unstructured":"Nuti, F., Franzmeyer, T., Henriques, J.F.: Extracting reward functions from diffusion models. In: Advances in Neural Information Processing Systems (NeurIPS) (2023)"},{"key":"27_CR31","unstructured":"Parisi, S., Rajeswaran, A., Purushwalkam, S., Gupta, A.: The unsurprising effectiveness of pre-trained vision models for control. In: International Conference on Machine Learning (ICML) (2022)"},{"key":"27_CR32","doi-asserted-by":"crossref","unstructured":"Peng, X.B., Ma, Z., Abbeel, P., Levine, S., Kanazawa, A.: AMP: adversarial motion priors for stylized physics-based character control. ACM Trans. Graph. (ToG) (2021)","DOI":"10.1145\/3476576.3476723"},{"key":"27_CR33","unstructured":"Pertsch, K., Lee, Y., Lim, J.: Accelerating reinforcement learning with learned skill priors. In: Conference on Robot Learning (CoRL) (2021)"},{"key":"27_CR34","doi-asserted-by":"crossref","unstructured":"Radosavovic, I., Wang, X., Pinto, L., Malik, J.: State-only imitation learning for dexterous manipulation. In: IEEE\/RSJ International Conference on Intelligent Robots and Systems (IROS) (2021)","DOI":"10.1109\/IROS51168.2021.9636557"},{"key":"27_CR35","doi-asserted-by":"crossref","unstructured":"Rajeswaran, A., Kumar, V., Gupta, A., Schulman, J., Todorov, E., Levine, S.: Learning complex dexterous manipulation with deep reinforcement learning and demonstrations. In: Robotics: Science and Systems (RSS) (2018)","DOI":"10.15607\/RSS.2018.XIV.049"},{"key":"27_CR36","unstructured":"Rao, D., et\u00a0al.: Learning transferable motor skills with hierarchical latent mixture policies. In: International Conference on Learning Representations (ICLR) (2021)"},{"key":"27_CR37","doi-asserted-by":"crossref","unstructured":"Sermanet, P., et al.: Time-contrastive networks: Self-supervised learning from video. In: IEEE International Conference on Robotics and Automation (ICRA) (2017)","DOI":"10.1109\/CVPRW.2017.69"},{"key":"27_CR38","doi-asserted-by":"crossref","unstructured":"Sermanet, P., Xu, K., Levine, S.: Unsupervised perceptual rewards for imitation learning. arXiv preprint arXiv:1612.06699 (2016)","DOI":"10.15607\/RSS.2017.XIII.050"},{"key":"27_CR39","unstructured":"Singh, A., Liu, H., Zhou, G., Yu, A., Rhinehart, N., Levine, S.: Parrot: data-driven behavioral priors for reinforcement learning. In: International Conference on Learning Representations (ICLR) (2020)"},{"key":"27_CR40","doi-asserted-by":"crossref","unstructured":"Singh, A., Yang, L., Hartikainen, K., Finn, C., Levine, S.: End-to-end robotic reinforcement learning without reward engineering. arXiv preprint arXiv:1904.07854 (2019)","DOI":"10.15607\/RSS.2019.XV.073"},{"key":"27_CR41","unstructured":"Sohl-Dickstein, J., Weiss, E., Maheswaranathan, N., Ganguli, S.: Deep unsupervised learning using nonequilibrium thermodynamics. In: International Conference on Machine Learning (ICML) (2015)"},{"key":"27_CR42","unstructured":"Song, J., Meng, C., Ermon, S.: Denoising diffusion implicit models. In: International Conference on Learning Representations (ICLR) (2022)"},{"key":"27_CR43","unstructured":"Song, Y., Sohl-Dickstein, J., Kingma, D.P., Kumar, A., Ermon, S., Poole, B.: Score-based generative modeling through stochastic differential equations. arXiv preprint arXiv:2011.13456 (2020)"},{"key":"27_CR44","doi-asserted-by":"crossref","unstructured":"Torabi, F., Warnell, G., Stone, P.: Behavioral cloning from observation. In: International Joint Conference on Artificial Intelligence (IJCAI) (2018)","DOI":"10.24963\/ijcai.2018\/687"},{"key":"27_CR45","unstructured":"Torabi, F., Warnell, G., Stone, P.: Generative adversarial imitation from observation. arXiv preprint arXiv:1807.06158 (2018)"},{"key":"27_CR46","unstructured":"Wang, C., Luo, X., Ross, K.W., Li, D.: VRL3: a data-driven framework for visual deep reinforcement learning. In: Advances in Neural Information Processing Systems (NeurIPS) (2022)"},{"key":"27_CR47","unstructured":"Wang, Z., Hunt, J.J., Zhou, M.: Diffusion policies as an expressive policy class for offline reinforcement learning. In: International Conference on Learning Representations (ICLR) (2023)"},{"key":"27_CR48","unstructured":"Xiao, T., Radosavovic, I., Darrell, T., Malik, J.: Masked visual pre-training for motor control. arXiv preprint arXiv:2203.06173 (2022)"},{"key":"27_CR49","unstructured":"Yan, W., Zhang, Y., Abbeel, P., Srinivas, A.: VideoGPT: video generation using VQ-VAE and transformers. arXiv preprint arXiv:2104.10157 (2021)"},{"key":"27_CR50","unstructured":"Yarats, D., Fergus, R., Lazaric, A., Pinto, L.: Mastering visual continuous control: improved data-augmented reinforcement learning. In: International Conference on Learning Representations (ICLR) (2022)"},{"key":"27_CR51","unstructured":"Yu, T., et al.: Meta-world: a benchmark and evaluation for multi-task and meta reinforcement learning. In: Conference on Robot Learning (CoRL) (2020)"},{"key":"27_CR52","doi-asserted-by":"crossref","unstructured":"Yu, T., et\u00a0al.: Scaling robot learning with semantically imagined experience. arXiv preprint arXiv:2302.11550 (2023)","DOI":"10.15607\/RSS.2023.XIX.027"},{"key":"27_CR53","unstructured":"Yuan, Z., et al.: Pre-trained image encoder for generalizable visual reinforcement learning. In: Advances in Neural Information Processing Systems (NeurIPS) (2022)"},{"key":"27_CR54","unstructured":"Zakka, K., Zeng, A., Florence, P., Tompson, J., Bohg, J., Dwibedi, D.: XIRL: cross-embodiment inverse reinforcement learning. In: Conference on Robot Learning (CoRL) (2022)"},{"key":"27_CR55","doi-asserted-by":"crossref","unstructured":"Zhang, R., Isola, P., Efros, A.A., Shechtman, E., Wang, O.: The unreasonable effectiveness of deep features as a perceptual metric. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR) (2018)","DOI":"10.1109\/CVPR.2018.00068"},{"key":"27_CR56","unstructured":"Zheng, R., et al.: TACO: temporal latent action-driven contrastive loss for visual reinforcement learning. arXiv preprint arXiv:2306.13229 (2023)"},{"key":"27_CR57","unstructured":"Zhu, J.Y., et al.: Toward multimodal image-to-image translation. Advances in Neural Information Processing Systems (NeurIPS) (2017)"},{"key":"27_CR58","unstructured":"Zhu, Z., Lin, K., Dai, B., Zhou, J.: Off-policy imitation learning from observations. In: Advances in Neural Information Processing Systems (NeurIPS) (2020)"},{"key":"27_CR59","unstructured":"Ziebart, B.D., Maas, A.L., Bagnell, J.A., Dey, A.K., et\u00a0al.: Maximum entropy inverse reinforcement learning. In: Association for the Advancement of Artificial Intelligence (AAAI) (2008)"}],"container-title":["Lecture Notes in Computer Science","Computer Vision \u2013 ECCV 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-72946-1_27","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,10,1]],"date-time":"2024-10-01T19:07:57Z","timestamp":1727809677000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-72946-1_27"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,2]]},"ISBN":["9783031729454","9783031729461"],"references-count":59,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-72946-1_27","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,2]]},"assertion":[{"value":"2 October 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECCV","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Computer Vision","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Milan","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 September 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 October 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"18","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"eccv2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/eccv2024.ecva.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}