<?xml version="1.0" encoding="UTF-8"?><xml><records><record><source-app name="Biblio" version="7.x">Drupal-Biblio</source-app><ref-type>17</ref-type><contributors><authors><author><style face="normal" font="default" size="100%">Burnetas, AN</style></author><author><style face="normal" font="default" size="100%">Katehakis, M.N.</style></author></authors></contributors><titles><title><style face="normal" font="default" size="100%">Optimal adaptive policies for markov decision processes</style></title><secondary-title><style face="normal" font="default" size="100%">Mathematics of Operations Research</style></secondary-title></titles><keywords><keyword><style  face="normal" font="default" size="100%">Adaptive control systems</style></keyword><keyword><style  face="normal" font="default" size="100%">Decision theory</style></keyword><keyword><style  face="normal" font="default" size="100%">Dynamic programming</style></keyword><keyword><style  face="normal" font="default" size="100%">Estimation</style></keyword><keyword><style  face="normal" font="default" size="100%">Markov processes</style></keyword><keyword><style  face="normal" font="default" size="100%">Markovian decision process (MDP)</style></keyword><keyword><style  face="normal" font="default" size="100%">Multi armed bandit (MAB) problem</style></keyword><keyword><style  face="normal" font="default" size="100%">Optimal control systems</style></keyword><keyword><style  face="normal" font="default" size="100%">Process control</style></keyword><keyword><style  face="normal" font="default" size="100%">State space methods</style></keyword><keyword><style  face="normal" font="default" size="100%">Statistics</style></keyword></keywords><dates><year><style  face="normal" font="default" size="100%">1997</style></year></dates><urls><web-urls><url><style face="normal" font="default" size="100%">https://www.scopus.com/inward/record.uri?eid=2-s2.0-0031070051&amp;partnerID=40&amp;md5=a3d110a158d5a27a6b74fd9d054a049b</style></url></web-urls></urls><number><style face="normal" font="default" size="100%">1</style></number><volume><style face="normal" font="default" size="100%">22</style></volume><pages><style face="normal" font="default" size="100%">222-255</style></pages><language><style face="normal" font="default" size="100%">eng</style></language><abstract><style face="normal" font="default" size="100%">In this paper we consider the problem of adaptive control for Markov Decision Processes. We give the explicit form for a class of adaptive policies that possess optimal increase rate properties for the total expected finite horizon reward, under sufficient assumptions of finite state-action spaces and irreducibility of the transition law. A main feature of the proposed policies is that the choice of actions, at each state and time period, is based on indices that are inflations of the right-hand side of the estimated average reward optimality equations.</style></abstract><notes><style face="normal" font="default" size="100%">cited By 42</style></notes></record></records></xml>