@inproceedings{6177496e6c8d4405bfc6820dd09e0d4f,
title = "Q* Approximation Schemes for Batch Reinforcement Learning: A Theoretical Comparison",
abstract = "We prove performance guarantees of two algorithms for approximating Q? in batch reinforcement learning. Compared to classical iterative methods such as Fitted Q-Iteration-whose performance loss incurs quadratic dependence on horizon-these methods estimate (some forms of) the Bellman error and enjoy linear-in-horizon error propagation, a property established for the first time for algorithms that rely solely on batch data and output stationary policies. One of the algorithms uses a novel and explicit importance-weighting correction to overcome the infamous {\textquotedblleft}double sampling{\textquotedblright} difficulty in Bellman error estimation, and does not use any squared losses. Our analyses reveal its distinct characteristics and potential advantages compared to classical algorithms.",
author = "Tengyang Xie and Nan Jiang",
note = "Publisher Copyright: {\textcopyright} 2020 Proceedings of Machine Learning Research. All rights reserved.; 36th Conference on Uncertainty in Artificial Intelligence, UAI 2020 ; Conference date: 03-08-2020 Through 06-08-2020",
year = "2020",
language = "English (US)",
series = "Proceedings of Machine Learning Research",
publisher = "ML Research Press",
pages = "550--559",
editor = "Jonas Peters and David Sontag",
booktitle = "Proceedings of the 36th Conference on Uncertainty in Artificial Intelligence (UAI)",
}