Companion code to reproduce experiments from research articles on Statistical Reinforcement Learning
-
Saber, H., Pesquerel, F., Maillard, O.A. and Talebi, M.S., 2024, February. Logarithmic regret in communicating MDPs: Leveraging known dynamics with bandits. In Asian Conference on Machine Learning (pp. 1167-1182). PMLR.
@inproceedings{saber2024logarithmic, title={Logarithmic regret in communicating MDPs: Leveraging known dynamics with bandits}, author={Saber, Hassan and Pesquerel, Fabien and Maillard, Odalric-Ambrym and Talebi, Mohammad Sadegh}, booktitle={Asian Conference on Machine Learning}, pages={1167--1182}, year={2024}, organization={PMLR} } -
Pesquerel, F. and Maillard, O.A., 2022, November. IMED-RL: Regret optimal learning of ergodic Markov decision processes. In NeurIPS 2022-Thirty-sixth Conference on Neural Information Processing Systems.
@inproceedings{pesquerel2022imed, title={IMED-RL: Regret optimal learning of ergodic Markov decision processes}, author={Pesquerel, Fabien and Maillard, Odalric-Ambrym}, booktitle={NeurIPS 2022-Thirty-sixth Conference on Neural Information Processing Systems}, year={2022} } -
Bourel, H., Maillard, O. and Talebi, M.S., 2020, November. Tightening exploration in upper confidence reinforcement learning. In International Conference on Machine Learning (pp. 1056-1066). PMLR.
@inproceedings{bourel2020tightening, title={Tightening exploration in upper confidence reinforcement learning}, author={Bourel, Hippolyte and Maillard, Odalric and Talebi, Mohammad Sadegh}, booktitle={International Conference on Machine Learning}, pages={1056--1066}, year={2020}, organization={PMLR} }