{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,11]],"date-time":"2026-06-11T12:36:03Z","timestamp":1781181363500,"version":"3.54.1"},"reference-count":51,"publisher":"Springer Science and Business Media LLC","issue":"11","license":[{"start":{"date-parts":[[2025,8,14]],"date-time":"2025-08-14T00:00:00Z","timestamp":1755129600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,8,14]],"date-time":"2025-08-14T00:00:00Z","timestamp":1755129600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Knowl Inf Syst"],"published-print":{"date-parts":[[2025,11]]},"DOI":"10.1007\/s10115-025-02562-8","type":"journal-article","created":{"date-parts":[[2025,8,14]],"date-time":"2025-08-14T08:14:21Z","timestamp":1755159261000},"page":"11075-11094","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":3,"title":["Enhancing deep reinforcement learning for stock trading: a reward shaping approach via expert feedback"],"prefix":"10.1007","volume":"67","author":[{"given":"Arishi","family":"Orra","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Himanshu","family":"Choudhary","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ankit","family":"Sharma","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Manoj","family":"Thakur","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2025,8,14]]},"reference":[{"key":"2562_CR1","doi-asserted-by":"publisher","first-page":"142","DOI":"10.1016\/j.ins.2020.05.066","volume":"538","author":"X Wu","year":"2020","unstructured":"Wu X, Chen H, Wang J, Troiano L, Loia V, Fujita H (2020) Adaptive stock trading strategies with deep reinforcement learning methods. Inf Sci 538:142\u2013158","journal-title":"Inf Sci"},{"issue":"6","key":"2562_CR2","doi-asserted-by":"publisher","first-page":"1305","DOI":"10.1007\/s00607-019-00773-w","volume":"102","author":"Y Li","year":"2020","unstructured":"Li Y, Ni P, Chang V (2020) Application of deep reinforcement learning in stock trading strategies and stock forecasting. Computing 102(6):1305\u20131322","journal-title":"Computing"},{"key":"2562_CR3","unstructured":"Zhang Z, Zohren S, Roberts S (2019) Deep reinforcement learning for trading. arXiv preprint arXiv:1911.10107"},{"key":"2562_CR4","unstructured":"Graham B, Dodd DLF, Cottle S, et al (1934) Security analysis vol. 452"},{"key":"2562_CR5","unstructured":"Murphy JJ (1999) Technical analysis of the financial markets: a comprehensive guide to trading methods and applications"},{"key":"2562_CR6","unstructured":"Chan EP (2021) Quantitative trading: how to build your own algorithmic trading business"},{"key":"2562_CR7","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/j.jocs.2016.07.006","volume":"17","author":"D Kumar","year":"2016","unstructured":"Kumar D, Meghwani SS, Thakur M (2016) Proximal support vector machine based hybrid prediction models for trend forecasting in financial markets. J Comput Sci 17:1\u201313","journal-title":"J Comput Sci"},{"key":"2562_CR8","doi-asserted-by":"crossref","unstructured":"Orra A, Sahoo K, Choudhary H (2023) Machine learning-based hybrid models for trend forecasting in financial instruments. In: Soft computing for problem solving: Proceedings of the SocProS 2022, pp. 337\u2013353","DOI":"10.1007\/978-981-19-6525-8_26"},{"key":"2562_CR9","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2022.118581","volume":"211","author":"Y Han","year":"2023","unstructured":"Han Y, Kim J, Enke D (2023) A machine learning trading system for the stock market based on n-period min-max labeling using xgboost. Expert Syst Appl 211:118581","journal-title":"Expert Syst Appl"},{"issue":"1","key":"2562_CR10","doi-asserted-by":"publisher","first-page":"259","DOI":"10.1016\/j.eswa.2014.07.040","volume":"42","author":"J Patel","year":"2015","unstructured":"Patel J, Shah S, Thakkar P, Kotecha K (2015) Predicting stock and stock price index movement using trend deterministic data preparation and machine learning techniques. Expert Syst Appl 42(1):259\u2013268","journal-title":"Expert Syst Appl"},{"key":"2562_CR11","doi-asserted-by":"publisher","first-page":"273","DOI":"10.1016\/j.eswa.2019.03.029","volume":"129","author":"E Hoseinzade","year":"2019","unstructured":"Hoseinzade E, Haratizadeh S (2019) Cnnpred: Cnn-based stock market prediction using a diverse set of variables. Expert Syst Appl 129:273\u2013285","journal-title":"Expert Syst Appl"},{"key":"2562_CR12","doi-asserted-by":"publisher","first-page":"1168","DOI":"10.1016\/j.procs.2020.03.049","volume":"170","author":"A Moghar","year":"2020","unstructured":"Moghar A, Hamiche M (2020) Stock market prediction using lstm recurrent neural network. Procedia Comput Sci 170:1168\u20131173","journal-title":"Procedia Comput Sci"},{"key":"2562_CR13","doi-asserted-by":"crossref","unstructured":"Bhambu A (2023) Stock market prediction using deep learning techniques for short and long horizon. In: Soft computing for problem solving: Proceedings of the SocProS 2022, pp. 121\u2013135","DOI":"10.1007\/978-981-19-6525-8_11"},{"issue":"7","key":"2562_CR14","doi-asserted-by":"publisher","first-page":"2837","DOI":"10.1109\/TNNLS.2020.2997523","volume":"32","author":"A Tsantekidis","year":"2020","unstructured":"Tsantekidis A, Passalis N, Toufa A-S, Saitas-Zarkias K, Chairistanidis S, Tefas A (2020) Price trailing for financial trading using deep reinforcement learning. IEEE Trans Neural Netw Learn Syst 32(7):2837\u20132846","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"key":"2562_CR15","unstructured":"Sutton RS, Barto AG (1998) The reinforcement learning problem. Reinforcement learning: An introduction, 51\u201385"},{"key":"2562_CR16","unstructured":"Mnih V, Kavukcuoglu K, Silver D, Graves A, Antonoglou I, Wierstra D, Riedmiller M (2013) Playing atari with deep reinforcement learning. arXiv preprint arXiv:1312.5602"},{"key":"2562_CR17","doi-asserted-by":"crossref","unstructured":"Orra A, Bhambu A, Choudhary H, Thakur M (2024) Dynamic reinforced ensemble using bayesian optimization for stock trading. In: Proceedings of the 5th ACM international conference on AI in Finance, pp. 361\u2013369","DOI":"10.1145\/3677052.3698595"},{"key":"2562_CR18","unstructured":"Orra A, Bhambu A, Choudhary H, Thakur M, Natarajan S (2025) Deep reinforcement learning for investor-specific portfolio optimization: a volatility-guided asset selection approach. arXiv preprint arXiv:2505.03760"},{"issue":"1","key":"2562_CR19","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1007\/s44196-025-00875-8","volume":"18","author":"H Choudhary","year":"2025","unstructured":"Choudhary H, Orra A, Sahoo K, Thakur M (2025) Risk-adjusted deep reinforcement learning for portfolio optimization: a multi-reward approach. Int J Comput Intell Syst 18(1):1\u201319","journal-title":"Int J Comput Intell Syst"},{"issue":"4","key":"2562_CR20","doi-asserted-by":"publisher","first-page":"875","DOI":"10.1109\/72.935097","volume":"12","author":"J Moody","year":"2001","unstructured":"Moody J, Saffell M (2001) Learning to trade via direct reinforcement. IEEE Trans Neural Netw 12(4):875\u2013889","journal-title":"IEEE Trans Neural Netw"},{"key":"2562_CR21","unstructured":"Lu DW (2017) Agent inspired trading using recurrent reinforcement learning and lstm neural networks. arXiv preprint arXiv:1707.07338"},{"key":"2562_CR22","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2020.113761","volume":"163","author":"JB Chakole","year":"2021","unstructured":"Chakole JB, Kolhe MS, Mahapurush GD, Yadav A, Kurhekar MP (2021) A q-learning agent for automated trading in equity stock markets. Expert Syst Appl 163:113761","journal-title":"Expert Syst Appl"},{"issue":"2","key":"2562_CR23","doi-asserted-by":"publisher","first-page":"2452","DOI":"10.1007\/s10489-022-03606-0","volume":"53","author":"X Yu","year":"2023","unstructured":"Yu X, Wu W, Liao X, Han Y (2023) Dynamic stock-decision ensemble strategy based on deep reinforcement learning. Appl Intell 53(2):2452\u20132470","journal-title":"Appl Intell"},{"key":"2562_CR24","unstructured":"Liu X-Y, Xiong Z, Zhong S, Yang H, Walid A (2018) Practical deep reinforcement learning approach for stock trading. arXiv preprint arXiv:1811.07522"},{"key":"2562_CR25","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2023.121849","volume":"238","author":"L Avramelou","year":"2024","unstructured":"Avramelou L, Nousi P, Passalis N, Tefas A (2024) Deep reinforcement learning for financial trading using multi-modal features. Expert Syst Appl 238:121849","journal-title":"Expert Syst Appl"},{"key":"2562_CR26","doi-asserted-by":"crossref","unstructured":"Rodinos G, Nousi P, Passalis N, Tefas A (2023) A sharpe ratio based reward scheme in deep reinforcement learning for financial trading. In: IFIP international conference on artificial intelligence applications and innovations, pp. 15\u201323. Springer","DOI":"10.1007\/978-3-031-34111-3_2"},{"issue":"5\u20136","key":"2562_CR27","doi-asserted-by":"publisher","first-page":"441","DOI":"10.1002\/(SICI)1099-131X(1998090)17:5\/6<441::AID-FOR707>3.0.CO;2-#","volume":"17","author":"J Moody","year":"1998","unstructured":"Moody J, Wu L, Liao Y, Saffell M (1998) Performance functions and reinforcement learning for trading systems and portfolios. J Forecast 17(5\u20136):441\u2013470","journal-title":"J Forecast"},{"key":"2562_CR28","doi-asserted-by":"publisher","first-page":"93564","DOI":"10.1109\/ACCESS.2022.3203697","volume":"10","author":"T Kabbani","year":"2022","unstructured":"Kabbani T, Duman E (2022) Deep reinforcement learning approach for trading automation in the stock market. IEEE Access 10:93564\u201393574","journal-title":"IEEE Access"},{"key":"2562_CR29","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2023.120939","volume":"234","author":"ZJ Ye","year":"2023","unstructured":"Ye ZJ, Schuller BW (2023) Human-aligned trading by imitative multi-loss reinforcement learning. Expert Syst Appl 234:120939","journal-title":"Expert Syst Appl"},{"key":"2562_CR30","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2023.122581","volume":"240","author":"Y Huang","year":"2024","unstructured":"Huang Y, Wan X, Zhang L, Lu X (2024) A novel deep reinforcement learning framework with bilstm-attention networks for algorithmic trading. Expert Syst Appl 240:122581","journal-title":"Expert Syst Appl"},{"key":"2562_CR31","doi-asserted-by":"crossref","unstructured":"Yang H, Liu X-Y, Zhong S, Walid A (2020) Deep reinforcement learning for automated stock trading: An ensemble strategy. In: Proceedings of the First ACM international conference on AI in finance, pp. 1\u20138","DOI":"10.1145\/3383455.3422540"},{"key":"2562_CR32","doi-asserted-by":"crossref","unstructured":"Azhikodan AR, Bhat AG, Jadhav MV (2019) Stock trading bot using deep reinforcement learning. In: Innovations in computer science and engineering: proceedings of the Fifth ICICSE 2017, pp. 41\u201349. Springer","DOI":"10.1007\/978-981-10-8201-6_5"},{"key":"2562_CR33","doi-asserted-by":"publisher","first-page":"267","DOI":"10.1016\/j.eswa.2017.06.023","volume":"87","author":"S Almahdi","year":"2017","unstructured":"Almahdi S, Yang SY (2017) An adaptive portfolio trading system: a risk-return portfolio optimization using recurrent reinforcement learning with expected maximum drawdown. Expert Syst Appl 87:267\u2013279","journal-title":"Expert Syst Appl"},{"key":"2562_CR34","first-page":"643","volume":"35","author":"Z Wang","year":"2021","unstructured":"Wang Z, Huang B, Tu S, Zhang K, Xu L (2021) Deeptrader: a deep reinforcement learning approach for risk-return balanced portfolio management with market conditions embedding. Proc AAAI Conf Artif Intell 35:643\u2013650","journal-title":"Proc AAAI Conf Artif Intell"},{"key":"2562_CR35","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2023.120297","volume":"227","author":"SJ Yoo","year":"2023","unstructured":"Yoo SJ, Gu YH et al (2023) Safety aarl: weight adjustment for reinforcement-learning-based safety dynamic asset allocation strategies. Expert Syst Appl 227:120297","journal-title":"Expert Syst Appl"},{"key":"2562_CR36","unstructured":"Schulman J, Wolski F, Dhariwal P, Radford A, Klimov O (2017) Proximal policy optimization algorithms. arXiv preprint arXiv:1707.06347"},{"key":"2562_CR37","doi-asserted-by":"crossref","unstructured":"Liu X-Y, Yang H, Chen Q, Zhang R, Yang L, Xiao B, Wang CD (2020) Finrl: A deep reinforcement learning library for automated stock trading in quantitative finance. arXiv preprint arXiv:2011.09607","DOI":"10.2139\/ssrn.3737257"},{"key":"2562_CR38","unstructured":"Pring MJ (2014) Technical analysis explained: the successful investor\u2019s guide to spotting investment trends and turning points"},{"issue":"5","key":"2562_CR39","doi-asserted-by":"publisher","first-page":"5311","DOI":"10.1016\/j.eswa.2010.10.027","volume":"38","author":"Y Kara","year":"2011","unstructured":"Kara Y, Boyacioglu MA, Baykan \u00d6K (2011) Predicting direction of stock price index movement using artificial neural networks and support vector machines: the sample of the istanbul stock exchange. Expert Syst Appl 38(5):5311\u20135319","journal-title":"Expert Syst Appl"},{"key":"2562_CR40","doi-asserted-by":"publisher","first-page":"337","DOI":"10.1016\/j.asoc.2018.03.006","volume":"67","author":"M Thakur","year":"2018","unstructured":"Thakur M, Kumar D (2018) A hybrid financial trading support system using multi-category classifiers and random forest. Appl Soft Comput 67:337\u2013349","journal-title":"Appl Soft Comput"},{"key":"2562_CR41","doi-asserted-by":"publisher","first-page":"170","DOI":"10.1016\/j.asoc.2016.01.048","volume":"43","author":"M Ozturk","year":"2016","unstructured":"Ozturk M, Toroslu IH, Fidan G (2016) Heuristic based trading system on forex data using technical indicator rules. Appl Soft Comput 43:170\u2013186","journal-title":"Appl Soft Comput"},{"key":"2562_CR42","unstructured":"Murphy JJ (2009) The visual investor: how to spot market trends"},{"key":"2562_CR43","unstructured":"Dahlquist JR, Kirkpatrick II CD (2010) Technical analysis: the complete resource for financial market technicians"},{"issue":"3","key":"2562_CR44","doi-asserted-by":"publisher","first-page":"653","DOI":"10.1109\/TNNLS.2016.2522401","volume":"28","author":"Y Deng","year":"2016","unstructured":"Deng Y, Bao F, Kong Y, Ren Z, Dai Q (2016) Deep direct reinforcement learning for financial signal representation and trading. IEEE Trans Neural Netw Learn Syst 28(3):653\u2013664","journal-title":"IEEE Trans Neural Netw Learn Syst"},{"key":"2562_CR45","doi-asserted-by":"crossref","unstructured":"Dabney W, Rowland M, Bellemare M, Munos R (2018) Distributional reinforcement learning with quantile regression. In: Proceedings of the AAAI conference on artificial intelligence, vol. 32","DOI":"10.1609\/aaai.v32i1.11791"},{"key":"2562_CR46","unstructured":"Snoek J, Larochelle H, Adams RP (2012) Practical bayesian optimization of machine learning algorithms. Adv Neural Inf Proc Syst 25"},{"key":"2562_CR47","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2020.113456","volume":"156","author":"F Soleymani","year":"2020","unstructured":"Soleymani F, Paquet E (2020) Financial portfolio optimization with online deep reinforcement learning and restricted stacked autoencoder\u2014deepbreath. Expert Syst Appl 156:113456","journal-title":"Expert Syst Appl"},{"key":"2562_CR48","unstructured":"Malkiel BG (1999) A random walk down wall street: including a life-cycle guide to personal investing"},{"issue":"2","key":"2562_CR49","doi-asserted-by":"publisher","first-page":"469","DOI":"10.1111\/j.1540-6261.1991.tb02669.x","volume":"46","author":"HM Markowitz","year":"1991","unstructured":"Markowitz HM (1991) Foundations of portfolio theory. J Financ 46(2):469\u2013477","journal-title":"J Financ"},{"key":"2562_CR50","doi-asserted-by":"crossref","unstructured":"Liu X-Y, Yang H, Gao J, Wang CD (2021) Finrl: Deep reinforcement learning framework to automate trading in quantitative finance. In: Proceedings of the second ACM international conference on AI in finance, pp. 1\u20139","DOI":"10.1145\/3490354.3494366"},{"key":"2562_CR51","doi-asserted-by":"crossref","unstructured":"Sun S, Xue W, Wang R, He X, Zhu J, Li J, An B (2022) Deepscalper: a risk-aware reinforcement learning framework to capture fleeting intraday trading opportunities. In: Proceedings of the 31st ACM international conference on information & knowledge management, pp. 1858\u20131867","DOI":"10.1145\/3511808.3557283"}],"container-title":["Knowledge and Information Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10115-025-02562-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10115-025-02562-8\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10115-025-02562-8.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,7]],"date-time":"2025-11-07T15:51:48Z","timestamp":1762530708000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10115-025-02562-8"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,8,14]]},"references-count":51,"journal-issue":{"issue":"11","published-print":{"date-parts":[[2025,11]]}},"alternative-id":["2562"],"URL":"https:\/\/doi.org\/10.1007\/s10115-025-02562-8","relation":{},"ISSN":["0219-1377","0219-3116"],"issn-type":[{"value":"0219-1377","type":"print"},{"value":"0219-3116","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,8,14]]},"assertion":[{"value":"5 October 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 July 2025","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 July 2025","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"14 August 2025","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no Conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}