ICML 2018oral67 citations
Learning to Explore via Meta-Policy Gradient
Tianbing Xu, Qiang Liu, Liang Zhao, Jian Peng
Abstract
The performance of off-policy learning, including deep Q-learning and deep deterministic policy gradient (DDPG), critically depends on the choice of the exploration policy. Existing exploration methods are mostly based on adding noise to the on-going actor policy and can only explore
BibTeX
@InProceedings{pmlr-v80-xu18d,
title = {Learning to Explore via Meta-Policy Gradient},
author = {Xu, Tianbing and Liu, Qiang and Zhao, Liang and Peng, Jian},
booktitle = {Proceedings of the 35th International Conference on Machine Learning},
pages = {5463--5472},
year = {2018},
editor = {Dy, Jennifer and Krause, Andreas},
volume = {80},
series = {Proceedings of Machine Learning Research},
month = {10--15 Jul},
publisher = {PMLR},
pdf = {http://proceedings.mlr.press/v80/xu18d/xu18d.pdf},
url = {https://proceedings.mlr.press/v80/xu18d.html},
abstract = {The performance of off-policy learning, including deep Q-learning and deep deterministic policy gradient (DDPG), critically depends on the choice of the exploration policy. Existing exploration methods are mostly based on adding noise to the on-going actor policy and can only explore