@inproceedings{d86a39dff4244ab2b7dcced9128b1333,
title = "Mean-variance optimization in Markov decision processes",
abstract = "We consider finite horizon Markov decision processes under performance measures that involve both the mean and the variance of the cumulative reward. We show that either randomized or history-based policies can improve performance. We prove that the complexity of computing a policy that maximizes the mean reward under a variance constraint is NP-hard for some cases, and strongly NP-hard for others. We finally offer pseudopolynomial exact and approximation algorithms.",
author = "Shie Mannor and Tsitsiklis, \{John N.\}",
year = "2011",
language = "الإنجليزيّة",
isbn = "9781450306195",
series = "Proceedings of the 28th International Conference on Machine Learning, ICML 2011",
pages = "177--184",
booktitle = "Proceedings of the 28th International Conference on Machine Learning, ICML 2011",
note = "28th International Conference on Machine Learning, ICML 2011 ; Conference date: 28-06-2011 Through 02-07-2011",
}