@inproceedings{919e909236d546fabd3e7e8521885314,
title = "Hindsight Balanced Reward Shaping",
abstract = "Sparse rewards is a tricky problem in reinforcement learning and reward shaping is commonly used to solve the problem of sparse rewards in specific tasks, but it often requires priori knowledge and manually designing rewards, which are costly in many cases. Hindsight experience replay (HER) solves the problem of sparse rewards in multi-goal scenarios by replacing the goal of a failed trajectory with a virtual goal. Our method integrates the ideas of reward shaping and HER, which has two advantages: First, it can automatically perform reward shaping without manually-designed reward functions; Second, it can solve the problem arising from the use of virtual goals in HER. Experiment results show our method can significantly improve the performance in both Bit-Flipping environment and Mujoco environment.",
keywords = "hindsight experience replay, reinforcement learning, reward shaping, sparse reward",
author = "Mengxuan Shao and Feng Jiang and Shaohui Liu and Kun Han and Debin Zhao",
note = "Publisher Copyright: {\textcopyright} 2023, The Author(s), under exclusive license to Springer Nature Singapore Pte Ltd.; 29th International Conference on Neural Information Processing, ICONIP 2022 ; Conference date: 22-11-2022 Through 26-11-2022",
year = "2023",
doi = "10.1007/978-981-99-1642-9\_42",
language = "英语",
isbn = "9789819916412",
series = "Communications in Computer and Information Science",
publisher = "Springer Science and Business Media Deutschland GmbH",
pages = "492--503",
editor = "Mohammad Tanveer and Sonali Agarwal and Seiichi Ozawa and Asif Ekbal and Adam Jatowt",
booktitle = "Neural Information Processing - 29th International Conference, ICONIP 2022, Proceedings",
address = "德国",
}