{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,26]],"date-time":"2025-03-26T05:23:08Z","timestamp":1742966588377,"version":"3.40.3"},"publisher-location":"Cham","reference-count":34,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031475450"},{"type":"electronic","value":"9783031475467"}],"license":[{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023]]},"DOI":"10.1007\/978-3-031-47546-7_16","type":"book-chapter","created":{"date-parts":[[2023,11,2]],"date-time":"2023-11-02T00:03:15Z","timestamp":1698883395000},"page":"231-244","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Mastering the\u00a0Card Game of\u00a0Jaipur Through Zero-Knowledge Self-Play Reinforcement Learning and\u00a0Action Masks"],"prefix":"10.1007","author":[{"given":"Cristina","family":"Cutajar","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8274-6177","authenticated-orcid":false,"given":"Josef","family":"Bajada","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,11,2]]},"reference":[{"key":"16_CR1","unstructured":"Dulac-Arnold, G., et al.: Deep reinforcement learning in large discrete action spaces. arXiv preprint arXiv:1512.07679 (2015)"},{"key":"16_CR2","doi-asserted-by":"publisher","first-page":"e1123","DOI":"10.7717\/peerj-cs.1123","volume":"8","author":"K Fujita","year":"2022","unstructured":"Fujita, K.: AlphaDDA: strategies for adjusting the playing strength of a fully trained AlphaZero system to a suitable human training partner. PeerJ Comput. Sci. 8, e1123 (2022)","journal-title":"PeerJ Comput. Sci."},{"key":"16_CR3","unstructured":"Ghory, I.: Reinforcement learning in board games. Technical report 105, Department of Computer Science, University of Bristol (2004)"},{"key":"16_CR4","unstructured":"van Hasselt, H.: Double Q-learning. In: Advances in Neural Information Processing Systems, vol. 23 (2010)"},{"key":"16_CR5","doi-asserted-by":"crossref","unstructured":"van Hasselt, H., Guez, A., Silver, D.: Deep reinforcement learning with double Q-learning. In: Proceedings of the 30th AAAI Conference on Artificial Intelligence (AAAI 2016) (2016)","DOI":"10.1609\/aaai.v30i1.10295"},{"key":"16_CR6","unstructured":"Huang, S., Kanervisto, A., Raffin, A., Wang, W., Onta\u00f1\u00f3n, S., Dossa, R.F.J.: A2C is a special case of PPO. arXiv preprint arXiv:2205.09123 (2022)"},{"key":"16_CR7","doi-asserted-by":"crossref","unstructured":"Justesen, N., Uth, L.M., Jakobsen, C., Moore, P.D., Togelius, J., Risi, S.: Blood bowl: a new board game challenge and competition for AI. In: 2019 IEEE Conference on Games (CoG), pp. 1\u20138. IEEE (2019)","DOI":"10.1109\/CIG.2019.8848063"},{"key":"16_CR8","doi-asserted-by":"crossref","unstructured":"Kanervisto, A., Scheller, C., Hautam\u00e4ki, V.: Action space shaping in deep reinforcement learning. In: 2020 IEEE Conference on Games (CoG), pp. 479\u2013486. IEEE (2020)","DOI":"10.1109\/CoG47356.2020.9231687"},{"key":"16_CR9","unstructured":"Karagiannakos, S.: The idea behind actor-critics and how A2C and A3C improve them (2018). https:\/\/theaisummer.com\/Actor_critics"},{"key":"16_CR10","doi-asserted-by":"crossref","unstructured":"Karunakaran, D., Worrall, S., Nebot, E.: Efficient statistical validation with edge cases to evaluate highly automated vehicles. In: 2020 IEEE 23rd International Conference on Intelligent Transportation Systems (ITSC), pp. 1\u20138. IEEE (2020)","DOI":"10.1109\/ITSC45102.2020.9294590"},{"key":"16_CR11","unstructured":"Konen, W.: Reinforcement learning for board games: the temporal difference algorithm. Technical report, Research Center CIOP (Computational Intelligence, Optimization and Data Mining), TH K\u00f6ln-Cologne University of Applied Sciences (2015)"},{"key":"16_CR12","unstructured":"Liu, J., Hou, P., Mu, L., Yu, Y., Huang, C.: Elements of effective deep reinforcement learning towards tactical driving decision making. arXiv preprint arXiv:1802.00332 (2018)"},{"key":"16_CR13","doi-asserted-by":"publisher","first-page":"147","DOI":"10.1016\/j.ijar.2021.12.004","volume":"142","author":"Y Liu","year":"2022","unstructured":"Liu, Y., Zheng, J., Chang, F.: Learning and planning in partially observable environments without prior domain knowledge. Int. J. Approximate Reasoning 142, 147\u2013160 (2022). https:\/\/doi.org\/10.1016\/j.ijar.2021.12.004","journal-title":"Int. J. Approximate Reasoning"},{"key":"16_CR14","unstructured":"Mnih, V., et al.: Asynchronous methods for deep reinforcement learning. In: International Conference on Machine Learning, pp. 1928\u20131937. PMLR (2016)"},{"key":"16_CR15","unstructured":"Mnih, V., et al.: Playing atari with deep reinforcement learning. arXiv preprint arXiv:1312.5602 (2013)"},{"issue":"7540","key":"16_CR16","doi-asserted-by":"publisher","first-page":"529","DOI":"10.1038\/nature14236","volume":"518","author":"V Mnih","year":"2015","unstructured":"Mnih, V., et al.: Human-level control through deep reinforcement learning. Nature 518(7540), 529\u2013533 (2015)","journal-title":"Nature"},{"key":"16_CR17","doi-asserted-by":"publisher","DOI":"10.1007\/978-981-19-0638-1","volume-title":"Deep Reinforcement Learning","author":"A Plaat","year":"2022","unstructured":"Plaat, A.: Deep Reinforcement Learning. Springer, Singapore (2022). https:\/\/doi.org\/10.1007\/978-981-19-0638-1"},{"key":"16_CR18","unstructured":"Schulman, J., Levine, S., Abbeel, P., Jordan, M., Moritz, P.: Trust region policy optimization. In: International Conference on Machine Learning, pp. 1889\u20131897. PMLR (2015)"},{"key":"16_CR19","unstructured":"Schulman, J., Wolski, F., Dhariwal, P., Radford, A., Klimov, O.: Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347 (2017)"},{"key":"16_CR20","unstructured":"Silver, D., et al.: Mastering chess and shogi by self-play with a general reinforcement learning algorithm. arXiv preprint arXiv:1712.01815 (12 2017)"},{"issue":"6419","key":"16_CR21","doi-asserted-by":"publisher","first-page":"1140","DOI":"10.1126\/science.aar6404","volume":"362","author":"D Silver","year":"2018","unstructured":"Silver, D., et al.: A general reinforcement learning algorithm that masters chess, shogi, and go through self-play. Science 362(6419), 1140\u20131144 (2018). https:\/\/doi.org\/10.1126\/science.aar6404","journal-title":"Science"},{"key":"16_CR22","volume-title":"Reinforcement Learning: An Introduction","author":"RS Sutton","year":"2018","unstructured":"Sutton, R.S., Barto, A.G.: Reinforcement Learning: An Introduction, 2nd edn. MIT Press, Cambridge (2018)","edition":"2"},{"key":"16_CR23","doi-asserted-by":"publisher","first-page":"200","DOI":"10.1016\/j.icte.2020.05.003","volume":"6","author":"CY Tang","year":"2020","unstructured":"Tang, C.Y., Liu, C.H., Chen, W.K., You, S.D.: Implementing action mask in proximal policy optimization (PPO) algorithm. ICT Express 6, 200\u2013203 (2020). https:\/\/doi.org\/10.1016\/j.icte.2020.05.003","journal-title":"ICT Express"},{"key":"16_CR24","unstructured":"Wang, Z., Schaul, T., Hessel, M., Hasselt, H., Lanctot, M., Freitas, N.: Dueling network architectures for deep reinforcement learning. In: International Conference on Machine Learning, pp. 1995\u20132003. PMLR (2016)"},{"key":"16_CR25","unstructured":"Watkins, C.J.: Learning from delayed rewards. Ph.D. thesis, King\u2019s College, Cambridge United Kingdom (1989)"},{"key":"16_CR26","doi-asserted-by":"publisher","first-page":"279","DOI":"10.1007\/BF00992698","volume":"8","author":"CJ Watkins","year":"1992","unstructured":"Watkins, C.J., Dayan, P.: Q-learning. Mach. Learn. 8, 279\u2013292 (1992)","journal-title":"Mach. Learn."},{"key":"16_CR27","unstructured":"Wiering, M.A., Patist, J.P., Mannen, H.: Learning to play board games using temporal difference methods. Technical report UU-CS-2005-048, Utrecht University (2005)"},{"key":"16_CR28","unstructured":"Wu, Y., Mansimov, E., Grosse, R.B., Liao, S., Ba, J.: Scalable trust-region method for deep reinforcement learning using kronecker-factored approximation. In: Advances in Neural Information Processing Systems, vol. 30 (2017)"},{"key":"16_CR29","doi-asserted-by":"crossref","unstructured":"Yao, Z., et al.: Towards modern card games with large-scale action spaces through action representation. In: 2022 IEEE Conference on Games (CoG), pp. 576\u2013579. IEEE (2022)","DOI":"10.1109\/CoG51982.2022.9893589"},{"key":"16_CR30","doi-asserted-by":"crossref","unstructured":"Ye, D., et al.: Mastering complex control in MOBA games with deep reinforcement learning. In: Proceedings of the 34th AAAI Conference on Artificial Intelligence (AAAI-20), pp. 6672\u20136679 (2020)","DOI":"10.1609\/aaai.v34i04.6144"},{"key":"16_CR31","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/s11633-022-1384-6","volume":"20","author":"QY Yin","year":"2023","unstructured":"Yin, Q.Y., et al.: Ai in human-computer gaming: techniques, challenges and opportunities. Mach. Intell. Res. 20, 1\u201319 (2023)","journal-title":"Mach. Intell. Res."},{"key":"16_CR32","first-page":"24611","volume":"35","author":"C Yu","year":"2022","unstructured":"Yu, C., et al.: The surprising effectiveness of PPO in cooperative multi-agent games. Adv. Neural. Inf. Process. Syst. 35, 24611\u201324624 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"16_CR33","unstructured":"Zahavy, T., Haroush, M., Merlis, N., Mankowitz, D.J., Mannor, S.: Learn what not to learn: action elimination with deep reinforcement learning. In: Advances in Neural Information Processing Systems, vol. 31 (2018)"},{"key":"16_CR34","unstructured":"Zha, D., et al.: Douzero: Mastering doudizhu with self-play deep reinforcement learning. In: International Conference on Machine Learning, pp. 12333\u201312344. PMLR (2021)"}],"container-title":["Lecture Notes in Computer Science","AIxIA 2023 \u2013 Advances in Artificial Intelligence"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-47546-7_16","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,11,2]],"date-time":"2023-11-02T00:13:25Z","timestamp":1698884005000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-47546-7_16"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023]]},"ISBN":["9783031475450","9783031475467"],"references-count":34,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-47546-7_16","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2023]]},"assertion":[{"value":"2 November 2023","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"AIxIA","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference of the Italian Association for Artificial Intelligence","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Rome","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2023","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"6 November 2023","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"9 November 2023","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"22","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"aiia2023","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"http:\/\/www.aixia2023.cnr.it\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Single-blind","order":1,"name":"type","label":"Type","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Easychair.org","order":2,"name":"conference_management_system","label":"Conference Management System","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"53","order":3,"name":"number_of_submissions_sent_for_review","label":"Number of Submissions Sent for Review","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"33","order":4,"name":"number_of_full_papers_accepted","label":"Number of Full Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"0","order":5,"name":"number_of_short_papers_accepted","label":"Number of Short Papers Accepted","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"62% - The value is computed by the equation \"Number of Full Papers Accepted \/ Number of Submissions Sent for Review * 100\" and then rounded to a whole number.","order":6,"name":"acceptance_rate_of_full_papers","label":"Acceptance Rate of Full Papers","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":7,"name":"average_number_of_reviews_per_paper","label":"Average Number of Reviews per Paper","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"3","order":8,"name":"average_number_of_papers_per_reviewer","label":"Average Number of Papers per Reviewer","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"Yes","order":9,"name":"external_reviewers_involved","label":"External Reviewers Involved","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}},{"value":"20 external reviewers.","order":10,"name":"additional_info_on_review_process","label":"Additional Info on Review Process","group":{"name":"ConfEventPeerReviewInformation","label":"Peer Review Information (provided by the conference organizers)"}}]}}