AISTATS 2016poster19 citations
A Fast and Reliable Policy Improvement Algorithm
Yasin Abbasi-Yadkori, Peter L. Bartlett, Stephen J. Wright
Abstract
We introduce a simple, efficient method that improves stochastic policies for Markov decision processes. The computational complexity is the same as that of the value estimation problem. We prove that when the value estimation error is small, this method gives an improvement in performance that increases with certain variance properties of the initial policy and transition dynamics. Performance in numerical experiments compares favorably with previous policy improvement algorithms.
BibTeX
@InProceedings{pmlr-v51-abbasi-yadkori16,
title = {A Fast and Reliable Policy Improvement Algorithm},
author = {Abbasi-Yadkori, Yasin and Bartlett, Peter L. and Wright, Stephen J.},
booktitle = {Proceedings of the 19th International Conference on Artificial Intelligence and Statistics},
pages = {1338--1346},
year = {2016},
editor = {Gretton, Arthur and Robert, Christian C.},
volume = {51},
series = {Proceedings of Machine Learning Research},
address = {Cadiz, Spain},
month = {09--11 May},
publisher = {PMLR},
pdf = {http://proceedings.mlr.press/v51/abbasi-yadkori16.pdf},
url = {https://proceedings.mlr.press/v51/abbasi-yadkori16.html},
abstract = {We introduce a simple, efficient method that improves stochastic policies for Markov decision processes. The computational complexity is the same as that of the value estimation problem. We prove that when the value estimation error is small, this method gives an improvement in performance that increases with certain variance properties of the initial policy and transition dynamics. Performance in numerical experiments compares favorably with previous policy improvement algorithms.}
}