AISTATS 2016poster340 citations
A Linearly-Convergent Stochastic L-BFGS Algorithm
Philipp Moritz, Robert Nishihara, Michael Jordan
Abstract
We propose a new stochastic L-BFGS algorithm and prove a linear convergence rate for strongly convex and smooth functions. Our algorithm draws heavily from a recent stochastic variant of L-BFGS proposed in Byrd et al. (2014) as well as a recent approach to variance reduction for stochastic gradient descent from Johnson and Zhang (2013). We demonstrate experimentally that our algorithm performs well on large-scale convex and non-convex optimization problems, exhibiting linear convergence and rapidly solving the optimization problems to high levels of precision. Furthermore, we show that our algorithm performs well for a wide-range of step sizes, often differing by several orders of magnitude.
BibTeX
@InProceedings{pmlr-v51-moritz16,
title = {A Linearly-Convergent Stochastic L-BFGS Algorithm},
author = {Moritz, Philipp and Nishihara, Robert and Jordan, Michael},
booktitle = {Proceedings of the 19th International Conference on Artificial Intelligence and Statistics},
pages = {249--258},
year = {2016},
editor = {Gretton, Arthur and Robert, Christian C.},
volume = {51},
series = {Proceedings of Machine Learning Research},
address = {Cadiz, Spain},
month = {09--11 May},
publisher = {PMLR},
pdf = {http://proceedings.mlr.press/v51/moritz16.pdf},
url = {https://proceedings.mlr.press/v51/moritz16.html},
abstract = {We propose a new stochastic L-BFGS algorithm and prove a linear convergence rate for strongly convex and smooth functions. Our algorithm draws heavily from a recent stochastic variant of L-BFGS proposed in Byrd et al. (2014) as well as a recent approach to variance reduction for stochastic gradient descent from Johnson and Zhang (2013). We demonstrate experimentally that our algorithm performs well on large-scale convex and non-convex optimization problems, exhibiting linear convergence and rapidly solving the optimization problems to high levels of precision. Furthermore, we show that our algorithm performs well for a wide-range of step sizes, often differing by several orders of magnitude.}
}