{"id":{"repo_id":"uiuc","oai_identifier":"oai:www.ideals.illinois.edu:2142/101086"},"canonical_url":"https://search.dev.ndltd.org/etd/uiuc/oai:www.ideals.illinois.edu:2142/101086","repository":{"repo_id":"uiuc","name":"University of Illinois - Urbana-Champaign","base_url":"https://www.ideals.illinois.edu/oai-pmh"},"display":{"title":"Solving planning problems with deep reinforcement learning and tree search","abstract":"Deep reinforcement learning methods are capable of learning complex heuristics starting with no prior knowledge, but struggle in environments where the learning signal is sparse. In contrast, planning methods can discover the optimal path to a goal in the absence of external rewards, but often require a hand-crafted heuristic function to be effective. In this thesis, we describe a model-based reinforcement learning method that bridges the middle ground between these two approaches. When evaluated on the complex domain of Sokoban, the model-based method was found to be more performant, stable and sample-efficient than a model-free baseline.","abstract_html":"Deep reinforcement learning methods are capable of learning complex heuristics starting with no prior knowledge, but struggle in environments where the learning signal is sparse. In contrast, planning methods can discover the optimal path to a goal in the absence of external rewards, but often require a hand-crafted heuristic function to be effective. In this thesis, we describe a model-based reinforcement learning method that bridges the middle ground between these two approaches. When evaluated on the complex domain of Sokoban, the model-based method was found to be more performant, stable and sample-efficient than a model-free baseline.","abstract_has_math":false,"creators":["Ge, Victor"],"institution":"University of Illinois at Urbana-Champaign","degree_name":"M.S.","degree_level":"Thesis","degree_discipline":"Computer Science","degree_department":null,"school":null,"contributors":["Lazebnik, Svetlana"],"advisors":[],"committee_chairs":[],"committee_members":[],"year":2018,"date_issued":"2018-09-04T20:32:01Z","date_published":"2018-09-04T20:32:01Z","updated_at":"2026-07-22T22:24:38Z","subjects":["reinforcement learning","mcts","sokoban","a*","heuristic"],"languages":["en"],"rights":["Copyright 2018 Victor Ge"],"rights_urls":[],"identifier_entries":[]},"links":{"outbound_url":"http://hdl.handle.net/2142/101086","outbound_label":"Handle","outbound_source":"dc:identifier"},"metadata_groups":[{"id":"people","label":"People","entries":[{"key":"dc:contributor","label":"Contributor","values":["Lazebnik, Svetlana"]},{"key":"dc:creator","label":"Author","values":["Ge, Victor"]}]},{"id":"academic_context","label":"Academic Context","entries":[{"key":"dc:date","label":"Dc Date","values":["2018-09-04T20:32:01Z","2018-04-26","2018-05"]},{"key":"dc:type","label":"Dc Type","values":["text"]},{"key":"thesis:degree_discipline","label":"Discipline","values":["Computer Science"]},{"key":"thesis:degree_level","label":"Degree Level","values":["Thesis"]},{"key":"thesis:degree_name","label":"Degree Name","values":["M.S."]},{"key":"thesis:institution_name","label":"Thesis Institution Name","values":["University of Illinois at Urbana-Champaign"]}]},{"id":"subjects_keywords","label":"Subjects and Keywords","entries":[{"key":"dc:subject","label":"Dc Subject","values":["reinforcement learning","mcts","sokoban","a*","heuristic"]}]},{"id":"language_rights","label":"Language and Rights","entries":[{"key":"dc:language","label":"Dc Language","values":["en"]},{"key":"dc:rights","label":"Dc Rights","values":["Copyright 2018 Victor Ge"]}]},{"id":"identifiers","label":"Identifiers","entries":[{"key":"dc:identifier","label":"Identifier","values":["http://hdl.handle.net/2142/101086"]}]},{"id":"additional","label":"Additional Metadata","entries":[{"key":"dc:description","label":"Description","values":["Deep reinforcement learning methods are capable of learning complex heuristics starting with no prior knowledge, but struggle in environments where the learning signal is sparse. In contrast, planning methods can discover the optimal path to a goal in the absence of external rewards, but often require a hand-crafted heuristic function to be effective. In this thesis, we describe a model-based reinforcement learning method that bridges the middle ground between these two approaches. When evaluated on the complex domain of Sokoban, the model-based method was found to be more performant, stable and sample-efficient than a model-free baseline.","Submission original under an indefinite embargo labeled 'Open Access'. The submission was exported from vireo on 2018-08-31 without embargo terms","The student, Victor Ge, accepted the attached license on 2018-04-25 at 18:20.","The student, Victor Ge, submitted this Thesis for approval on 2018-04-25 at 18:28.","This Thesis was approved for publication on 2018-04-26 at 15:04.","DSpace SAF Submission Ingestion Package generated from Vireo submission #12504 on 2018-08-31 at 17:15:02","Made available in DSpace on 2018-09-04T20:32:01Z (GMT). No. of bitstreams: 2 GE-THESIS-2018.pdf: 671819 bytes, checksum: 161b3332ca985b02b599f786084fffd2 (MD5) LICENSE.txt: 4206 bytes, checksum: aa9b2beecc67250f9c9c384d5880b2d6 (MD5) Previous issue date: 2018-04-26"]},{"key":"dc:format","label":"Dc Format","values":["application/pdf"]},{"key":"dc:title","label":"Title","values":["Solving planning problems with deep reinforcement learning and tree search"]}]}],"canonical_facts":{"dc:contributor":["Lazebnik, Svetlana"],"dc:creator":["Ge, Victor"],"dc:date":["2018-09-04T20:32:01Z","2018-04-26","2018-05"],"dc:description":["Deep reinforcement learning methods are capable of learning complex heuristics starting with no prior knowledge, but struggle in environments where the learning signal is sparse. In contrast, planning methods can discover the optimal path to a goal in the absence of external rewards, but often require a hand-crafted heuristic function to be effective. In this thesis, we describe a model-based reinforcement learning method that bridges the middle ground between these two approaches. When evaluated on the complex domain of Sokoban, the model-based method was found to be more performant, stable and sample-efficient than a model-free baseline.","Submission original under an indefinite embargo labeled 'Open Access'. The submission was exported from vireo on 2018-08-31 without embargo terms","The student, Victor Ge, accepted the attached license on 2018-04-25 at 18:20.","The student, Victor Ge, submitted this Thesis for approval on 2018-04-25 at 18:28.","This Thesis was approved for publication on 2018-04-26 at 15:04.","DSpace SAF Submission Ingestion Package generated from Vireo submission #12504 on 2018-08-31 at 17:15:02","Made available in DSpace on 2018-09-04T20:32:01Z (GMT). No. of bitstreams: 2 GE-THESIS-2018.pdf: 671819 bytes, checksum: 161b3332ca985b02b599f786084fffd2 (MD5) LICENSE.txt: 4206 bytes, checksum: aa9b2beecc67250f9c9c384d5880b2d6 (MD5) Previous issue date: 2018-04-26"],"dc:format":["application/pdf"],"dc:identifier":["http://hdl.handle.net/2142/101086"],"dc:language":["en"],"dc:rights":["Copyright 2018 Victor Ge"],"dc:subject":["reinforcement learning","mcts","sokoban","a*","heuristic"],"dc:title":["Solving planning problems with deep reinforcement learning and tree search"],"dc:type":["text"],"thesis:degree_discipline":["Computer Science"],"thesis:degree_level":["Thesis"],"thesis:degree_name":["M.S."],"thesis:institution_name":["University of Illinois at Urbana-Champaign"]},"updated_at":"2026-07-22T22:24:38Z"}