{"id":{"repo_id":"uiuc","oai_identifier":"oai:www.ideals.illinois.edu:2142/124419"},"canonical_url":"https://search.dev.ndltd.org/etd/uiuc/oai:www.ideals.illinois.edu:2142/124419","repository":{"repo_id":"uiuc","name":"University of Illinois - Urbana-Champaign","base_url":"https://www.ideals.illinois.edu/oai-pmh"},"display":{"title":"The Role of Lookahead in Reinforcement Learning Algorithms","abstract":"Submission original under an indefinite embargo labeled 'Open Access'. The submission was exported from vireo on 2024-09-16 without embargo terms","abstract_html":"Submission original under an indefinite embargo labeled &#x27;Open Access&#x27;. The submission was exported from vireo on 2024-09-16 without embargo terms","abstract_has_math":false,"creators":["Winnicki, Anna"],"institution":"University of Illinois at Urbana-Champaign","degree_name":"Ph.D.","degree_level":"Dissertation","degree_discipline":"Electrical & Computer Engr","degree_department":null,"school":null,"contributors":["Srikant, R.","Hajek, Bruce","Wierman, Adam","Beck, Carolyn","Sowers, Richard"],"advisors":[],"committee_chairs":[],"committee_members":[],"year":2024,"date_issued":"2024-05","date_published":"2024-05","updated_at":"2026-07-22T22:25:00Z","subjects":["Reinforcement Learning","Markov Decision Processes"],"languages":["en","eng"],"rights":["Copyright 2024 Anna Winnicki"],"rights_urls":[],"identifier_entries":[]},"links":{"outbound_url":"https://hdl.handle.net/2142/124419","outbound_label":"Handle","outbound_source":"dc:identifier"},"metadata_groups":[{"id":"people","label":"People","entries":[{"key":"dc:contributor","label":"Contributor","values":["Srikant, R.","Hajek, Bruce","Wierman, Adam","Beck, Carolyn","Sowers, Richard"]},{"key":"dc:creator","label":"Author","values":["Winnicki, Anna"]}]},{"id":"academic_context","label":"Academic Context","entries":[{"key":"dc:date","label":"Dc Date","values":["2024-05","2024-04-26"]},{"key":"dc:type","label":"Dc Type","values":["text"]},{"key":"thesis:degree_discipline","label":"Discipline","values":["Electrical & Computer Engr"]},{"key":"thesis:degree_level","label":"Degree Level","values":["Dissertation"]},{"key":"thesis:degree_name","label":"Degree Name","values":["Ph.D."]},{"key":"thesis:institution_name","label":"Thesis Institution Name","values":["University of Illinois at Urbana-Champaign"]}]},{"id":"subjects_keywords","label":"Subjects and Keywords","entries":[{"key":"dc:subject","label":"Dc Subject","values":["Reinforcement Learning","Markov Decision Processes"]}]},{"id":"language_rights","label":"Language and Rights","entries":[{"key":"dc:language","label":"Dc Language","values":["en","eng"]},{"key":"dc:rights","label":"Dc Rights","values":["Copyright 2024 Anna Winnicki"]}]},{"id":"identifiers","label":"Identifiers","entries":[{"key":"dc:identifier","label":"Identifier","values":["https://hdl.handle.net/2142/124419"]}]},{"id":"additional","label":"Additional Metadata","entries":[{"key":"dc:description","label":"Description","values":["Submission original under an indefinite embargo labeled 'Open Access'. The submission was exported from vireo on 2024-09-16 without embargo terms","The student, Anna Winnicki, accepted the attached license on 2024-04-26 at 14:14.","The student, Anna Winnicki, submitted this Dissertation for approval on 2024-04-26 at 14:18.","This Dissertation was approved for publication on 2024-04-26 at 16:18.","DSpace SAF Submission Ingestion Package generated from Vireo submission #20667 on 2024-09-16 at 00:37:10","State of the art reinforcement learning (RL) algorithms such as AlphaZero use lookahead, which is typically implemented using Monte Carlo Tree Search (MCTS). As the name suggests, lookahead simply means looking ahead several steps when computing the policy to be used. The fact that an H-step lookahead provides an O(alpha^H), where alpha is the discount factor, approximate solution to the optimal policy is a somewhat trivial and well-known statement. What we have shown is a much stronger result: we have shown that lookahead leads to convergent learning algorithms while the same algorithms may diverge in the absence of lookahead. We have demonstrated these results for three different classes of RL algorithms: modified policy iteration with linear value function approximation [1], Monte Carlo with exploring starts [2], and policy iteration for zero-sum Markov games [3]. We have also shown that lookahead can be efficiently implemented in the widely studied class of linear MDPs [3]."]},{"key":"dc:format","label":"Dc Format","values":["application/pdf"]},{"key":"dc:title","label":"Title","values":["The Role of Lookahead in Reinforcement Learning Algorithms"]}]}],"canonical_facts":{"dc:contributor":["Srikant, R.","Hajek, Bruce","Wierman, Adam","Beck, Carolyn","Sowers, Richard"],"dc:creator":["Winnicki, Anna"],"dc:date":["2024-05","2024-04-26"],"dc:description":["Submission original under an indefinite embargo labeled 'Open Access'. The submission was exported from vireo on 2024-09-16 without embargo terms","The student, Anna Winnicki, accepted the attached license on 2024-04-26 at 14:14.","The student, Anna Winnicki, submitted this Dissertation for approval on 2024-04-26 at 14:18.","This Dissertation was approved for publication on 2024-04-26 at 16:18.","DSpace SAF Submission Ingestion Package generated from Vireo submission #20667 on 2024-09-16 at 00:37:10","State of the art reinforcement learning (RL) algorithms such as AlphaZero use lookahead, which is typically implemented using Monte Carlo Tree Search (MCTS). As the name suggests, lookahead simply means looking ahead several steps when computing the policy to be used. The fact that an H-step lookahead provides an O(alpha^H), where alpha is the discount factor, approximate solution to the optimal policy is a somewhat trivial and well-known statement. What we have shown is a much stronger result: we have shown that lookahead leads to convergent learning algorithms while the same algorithms may diverge in the absence of lookahead. We have demonstrated these results for three different classes of RL algorithms: modified policy iteration with linear value function approximation [1], Monte Carlo with exploring starts [2], and policy iteration for zero-sum Markov games [3]. We have also shown that lookahead can be efficiently implemented in the widely studied class of linear MDPs [3]."],"dc:format":["application/pdf"],"dc:identifier":["https://hdl.handle.net/2142/124419"],"dc:language":["en","eng"],"dc:rights":["Copyright 2024 Anna Winnicki"],"dc:subject":["Reinforcement Learning","Markov Decision Processes"],"dc:title":["The Role of Lookahead in Reinforcement Learning Algorithms"],"dc:type":["text"],"thesis:degree_discipline":["Electrical & Computer Engr"],"thesis:degree_level":["Dissertation"],"thesis:degree_name":["Ph.D."],"thesis:institution_name":["University of Illinois at Urbana-Champaign"]},"updated_at":"2026-07-22T22:25:00Z"}