UAI 2023poster4 citations
Loosely consistent emphatic temporal-difference learning
Jiamin He, Fengdi Che, Yi Wan, A. Rupam Mahmood
Abstract
There has been significant interest in searching for off-policy Temporal-Difference (TD) algorithms that find the same solution that would have been obtained in the on-policy regime. An important property of such algorithms is that their expected update has the same fixed point as that of On-policy TD($\lambda$), which we call
BibTeX
@InProceedings{pmlr-v216-he23a,
title = {Loosely consistent emphatic temporal-difference learning},
author = {He, Jiamin and Che, Fengdi and Wan, Yi and Mahmood, A. Rupam},
booktitle = {Proceedings of the Thirty-Ninth Conference on Uncertainty in Artificial Intelligence},
pages = {849--859},
year = {2023},
editor = {Evans, Robin J. and Shpitser, Ilya},
volume = {216},
series = {Proceedings of Machine Learning Research},
month = {31 Jul--04 Aug},
publisher = {PMLR},
pdf = {https://proceedings.mlr.press/v216/he23a/he23a.pdf},
url = {https://proceedings.mlr.press/v216/he23a.html},
abstract = {There has been significant interest in searching for off-policy Temporal-Difference (TD) algorithms that find the same solution that would have been obtained in the on-policy regime. An important property of such algorithms is that their expected update has the same fixed point as that of On-policy TD($\lambda$), which we call