@inproceedings{e6529f264b5f4df6a542b17dfe7c11c3,
title = "Quantile-Based Policy Optimization for Reinforcement Learning",
abstract = "Classical reinforcement learning (RL) aims to optimize the expected cumulative rewards. In this work, we consider the RL setting where the goal is to optimize the quantile of the cumulative rewards. We parameterize the policy controlling actions by neural networks and propose a novel policy gradient algorithm called Quantile-Based Policy Optimization (QPO) and its variant Quantile-Based Proximal Policy Optimization (QPPO) to solve deep RL problems with quantile objectives. QPO uses two coupled iterations running at different time scales for simultaneously estimating quantiles and policy parameters. Our numerical results demonstrate that the proposed algorithms outperform the existing baseline algorithms under the quantile criterion.",
author = "Jinyang Jiang and Yijie Peng and Jiaqiao Hu",
note = "Publisher Copyright: {\textcopyright} 2022 IEEE.; 2022 Winter Simulation Conference, WSC 2022 ; Conference date: 11-12-2022 Through 14-12-2022",
year = "2022",
doi = "10.1109/WSC57314.2022.10015456",
language = "English",
series = "Proceedings - Winter Simulation Conference",
publisher = "Institute of Electrical and Electronics Engineers Inc.",
pages = "2712--2723",
editor = "B. Feng and G. Pedrielli and Y. Peng and S. Shashaani and E. Song and C.G. Corlu and L.H. Lee and E.P. Chew and T. Roeder and P. Lendermann",
booktitle = "Proceedings of the 2022 Winter Simulation Conference, WSC 2022",
}