{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,24]],"date-time":"2026-03-24T18:58:59Z","timestamp":1774378739689,"version":"3.50.1"},"reference-count":15,"publisher":"The Open Journal","issue":"119","license":[{"start":{"date-parts":[[2026,3,24]],"date-time":"2026-03-24T00:00:00Z","timestamp":1774310400000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by\/4.0\/"},{"start":{"date-parts":[[2026,3,24]],"date-time":"2026-03-24T00:00:00Z","timestamp":1774310400000},"content-version":"am","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by\/4.0\/"},{"start":{"date-parts":[[2026,3,24]],"date-time":"2026-03-24T00:00:00Z","timestamp":1774310400000},"content-version":"tdm","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["JOSS"],"published-print":{"date-parts":[[2026,3,24]]},"DOI":"10.21105\/joss.09620","type":"journal-article","created":{"date-parts":[[2026,3,24]],"date-time":"2026-03-24T15:59:52Z","timestamp":1774367992000},"page":"9620","source":"Crossref","is-referenced-by-count":0,"title":["Crux.jl: Deep Reinforcement Learning in Julia"],"prefix":"10.21105","volume":"11","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-2403-454X","authenticated-orcid":false,"given":"Robert J.","family":"Moss","sequence":"first","affiliation":[{"name":"Stanford University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-4027-0473","authenticated-orcid":false,"given":"Anthony","family":"Corso","sequence":"additional","affiliation":[{"name":"Stanford University"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7238-9663","authenticated-orcid":false,"given":"Mykel J.","family":"Kochenderfer","sequence":"additional","affiliation":[{"name":"Stanford University"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"8722","reference":[{"issue":"1","key":"julia","doi-asserted-by":"publisher","DOI":"10.1137\/141000671","article-title":"Julia: A fresh approach to numerical computing","volume":"59","author":"Bezanson","year":"2017","unstructured":"Bezanson, J., Edelman, A., Karpinski, S., & Shah, V. B. (2017). Julia: A fresh approach to numerical computing. SIAM Review, 59(1), 65\u201398. https:\/\/doi.org\/10.1137\/141000671","journal-title":"SIAM Review"},{"issue":"26","key":"pomdps_jl","article-title":"POMDPs.jl: A framework for sequential decision making under uncertainty","volume":"18","author":"Egorov","year":"2017","unstructured":"Egorov, M., Sunberg, Z. N., Balaban, E., Wheeler, T. A., Gupta, J. K., & Kochenderfer, M. J. (2017). POMDPs.jl: A framework for sequential decision making under uncertainty. Journal of Machine Learning Research, 18(26), 1\u20135.","journal-title":"Journal of Machine Learning Research"},{"key":"trpo","article-title":"Trust region policy optimization","author":"Schulman","year":"2015","unstructured":"Schulman, J., Levine, S., Abbeel, P., Jordan, M., & Moritz, P. (2015). Trust region policy optimization. International Conference on Machine Learning (ICML), 1889\u20131897.","journal-title":"International Conference on Machine Learning (ICML)"},{"key":"ppo","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1707.06347","article-title":"Proximal policy optimization algorithms","author":"Schulman","year":"2017","unstructured":"Schulman, J., Wolski, F., Dhariwal, P., Radford, A., & Klimov, O. (2017). Proximal policy optimization algorithms. arXiv:1707.06347. https:\/\/doi.org\/10.48550\/arXiv.1707.06347","journal-title":"arXiv:1707.06347"},{"key":"zeng2025reinforcement","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2504.16045","article-title":"Reinforcement learning and metaheuristics for Feynman integral reduction","author":"Zeng","year":"2025","unstructured":"Zeng, M. (2025). Reinforcement learning and metaheuristics for Feynman integral reduction. arXiv:2504.16045. https:\/\/doi.org\/10.48550\/arXiv.2504.16045","journal-title":"arXiv:2504.16045"},{"key":"banerjee2024energy","doi-asserted-by":"publisher","DOI":"10.2514\/6.2024-4545","article-title":"Energy-optimized path planning for UAS in varying winds via reinforcement learning","author":"Banerjee","year":"2024","unstructured":"Banerjee, P., & Bradner, K. (2024). Energy-optimized path planning for UAS in varying winds via reinforcement learning. AIAA AVIATION FORUM and ASCEND. https:\/\/doi.org\/10.2514\/6.2024-4545","journal-title":"AIAA AVIATION FORUM and ASCEND"},{"key":"valbook","volume-title":"Algorithms for Validation","author":"Kochenderfer","year":"2026","unstructured":"Kochenderfer, M. J., Katz, S. M., Corso, A. L., & Moss, R. J. (2026). Algorithms for Validation. MIT Press."},{"issue":"25","key":"flux","doi-asserted-by":"publisher","DOI":"10.21105\/joss.00602","article-title":"Flux: Elegant machine learning with Julia","volume":"3","author":"Innes","year":"2018","unstructured":"Innes, M. (2018). Flux: Elegant machine learning with Julia. Journal of Open Source Software, 3(25), 602. https:\/\/doi.org\/10.21105\/joss.00602","journal-title":"Journal of Open Source Software"},{"issue":"7540","key":"dqn","doi-asserted-by":"publisher","DOI":"10.1038\/nature14236","article-title":"Human-level control through deep reinforcement learning","volume":"518","author":"Mnih","year":"2015","unstructured":"Mnih, V., Kavukcuoglu, K., Silver, D., & others. (2015). Human-level control through deep reinforcement learning. Nature, 518(7540), 529\u2013533. https:\/\/doi.org\/10.1038\/nature14236","journal-title":"Nature"},{"key":"td3","article-title":"Addressing function approximation error in actor-critic methods","author":"Fujimoto","year":"2018","unstructured":"Fujimoto, S., Hoof, H., & Meger, D. (2018). Addressing function approximation error in actor-critic methods. International Conference on Machine Learning (ICML), 1587\u20131596.","journal-title":"International conference on machine learning (ICML)"},{"key":"sac","article-title":"Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor","author":"Haarnoja","year":"2018","unstructured":"Haarnoja, T., Zhou, A., Abbeel, P., & Levine, S. (2018). Soft actor-critic: Off-policy maximum entropy deep reinforcement learning with a stochastic actor. International Conference on Machine Learning (ICML), 1861\u20131870.","journal-title":"International conference on machine learning (ICML)"},{"key":"gym","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2407.17032","article-title":"Gymnasium: A standard interface for reinforcement learning environments","author":"Towers","year":"2024","unstructured":"Towers, M., Kwiatkowski, A., Terry, J., Balis, J. U., De Cola, G., Deleu, T., Goul\u00e3o, M., Kallinteris, A., Krimmel, M., KG, A., & others. (2024). Gymnasium: A standard interface for reinforcement learning environments. arXiv:2407.17032. https:\/\/doi.org\/10.48550\/arXiv.2407.17032","journal-title":"arXiv:2407.17032"},{"issue":"3","key":"reinforce","doi-asserted-by":"publisher","DOI":"10.1007\/BF00992696","article-title":"Simple statistical gradient-following algorithms for connectionist reinforcement learning","volume":"8","author":"Williams","year":"1992","unstructured":"Williams, R. J. (1992). Simple statistical gradient-following algorithms for connectionist reinforcement learning. Machine Learning, 8(3), 229\u2013256. https:\/\/doi.org\/10.1007\/BF00992696","journal-title":"Machine Learning"},{"issue":"268","key":"sb3","article-title":"Stable-baselines3: Reliable reinforcement learning implementations","volume":"22","author":"Raffin","year":"2021","unstructured":"Raffin, A., Hill, A., Gleave, A., Kanervisto, A., Ernestus, M., & Dormann, N. (2021). Stable-baselines3: Reliable reinforcement learning implementations. Journal of Machine Learning Research, 22(268), 1\u20138.","journal-title":"Journal of Machine Learning Research"},{"key":"rllib","article-title":"RLlib: Abstractions for distributed reinforcement learning","author":"Liang","year":"2018","unstructured":"Liang, E., Liaw, R., Nishihara, R., Moritz, P., Fox, R., Goldberg, K., Gonzalez, J. E., Jordan, M. I., & Ion Stoica. (2018). RLlib: Abstractions for distributed reinforcement learning. International Conference on Machine Learning (ICML).","journal-title":"International conference on machine learning (ICML)"}],"container-title":["Journal of Open Source Software"],"original-title":[],"link":[{"URL":"https:\/\/joss.theoj.org\/papers\/10.21105\/joss.09620.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,3,24]],"date-time":"2026-03-24T15:59:55Z","timestamp":1774367995000},"score":1,"resource":{"primary":{"URL":"https:\/\/joss.theoj.org\/papers\/10.21105\/joss.09620"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3,24]]},"references-count":15,"journal-issue":{"issue":"119","published-online":{"date-parts":[[2026,3]]}},"alternative-id":["10.21105\/joss.09620"],"URL":"https:\/\/doi.org\/10.21105\/joss.09620","relation":{"has-review":[{"id-type":"uri","id":"https:\/\/github.com\/openjournals\/joss-reviews\/issues\/9620","asserted-by":"subject"}],"references":[{"id-type":"doi","id":"10.5281\/zenodo.19152815","asserted-by":"subject"}]},"ISSN":["2475-9066"],"issn-type":[{"value":"2475-9066","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,3,24]]}}}