@inproceedings{e6bd40e0b02e474b943d81cde2cd44aa,
title = "Periodic Guidance Learning",
abstract = "Tasks with periodic states are widespread in reality. However, Current reinforcement learning (RL) algorithms generally treat such tasks as non-periodic Markov decision process, which results in low exploration efficiency and misleading advantage estimation with high variance. This paper proposes periodic guidance learning (PGL), in which a pruned advantage estimation with lower variance is implemented. Meanwhile, based on periodic states, past good experiences are utilized for better exploration. Our algorithm is evaluated on periodic tasks in MuJoCo. The experimental results show PGL method improves exploration efficiency and outperforms baselines in various periodic tasks. The results also show that PGL achieves a smooth policy optimization. Further experiments on the agent's periodic behavior reveal the strong correlation between period length and the agents motion mode.",
keywords = "Exploitation-exploration, Periodic tasks, Reinforcement learning",
author = "Lipeng Wan and Xuguang Lan and Xuwei Song and Chuzhen Feng and Nanning Zheng",
note = "Publisher Copyright: {\textcopyright} 2020 IEEE.; 11th IEEE International Conference on Knowledge Graph, ICKG 2020 ; Conference date: 09-08-2020 Through 11-08-2020",
year = "2020",
month = aug,
doi = "10.1109/ICBK50248.2020.00021",
language = "英语",
series = "Proceedings - 11th IEEE International Conference on Knowledge Graph, ICKG 2020",
publisher = "Institute of Electrical and Electronics Engineers Inc.",
pages = "77--83",
editor = "Enhong Chen and Grigoris Antoniou and Xindong Wu and Vipin Kumar",
booktitle = "Proceedings - 11th IEEE International Conference on Knowledge Graph, ICKG 2020",
}