{"id":{"repo_id":"york","oai_identifier":"oai:yorkspace.library.yorku.ca:10315/41039"},"canonical_url":"https://search.dev.ndltd.org/etd/york/oai:yorkspace.library.yorku.ca:10315/41039","repository":{"repo_id":"york","name":"York University","base_url":"https://yorkspace.library.yorku.ca/oai/request"},"display":{"title":"Exploiting Reward Machines with Deep Reinforcement Learning in Continuous Action Domains","abstract":"Deep reinforcement learning can solve real-world robot control problems, such as autonomous driving and robotic arm manipulation. In deep reinforcement learning, an agent does not know the problem description and learns the optimal solution through trial-and-error. This method brings two major challenges when solving real-world problems: partial observability and learning efficiency. In this thesis, we address these two challenges and extend previous work. First, we use reward machines to address the problem of partial observability. Then, we focus on finding the existing cutting-edge deep reinforcement learning algorithms and integrating them with reward machines to enhance the learning efficiency. To test the performance of all the algorithms, we proposed a series of different tasks that can be used to mimic real-world robot control problems. Finally, based on the test results, we compare the performance of all the algorithms and analyze their advantages and disadvantages.","abstract_html":"Deep reinforcement learning can solve real-world robot control problems, such as autonomous driving and robotic arm manipulation. In deep reinforcement learning, an agent does not know the problem description and learns the optimal solution through trial-and-error. This method brings two major challenges when solving real-world problems: partial observability and learning efficiency. In this thesis, we address these two challenges and extend previous work. First, we use reward machines to address the problem of partial observability. Then, we focus on finding the existing cutting-edge deep reinforcement learning algorithms and integrating them with reward machines to enhance the learning efficiency. To test the performance of all the algorithms, we proposed a series of different tasks that can be used to mimic real-world robot control problems. Finally, based on the test results, we compare the performance of all the algorithms and analyze their advantages and disadvantages.","abstract_has_math":false,"creators":["Sun, Haolin"],"institution":null,"degree_name":null,"degree_level":null,"degree_discipline":null,"degree_department":null,"school":null,"contributors":[],"advisors":["Lesperance, Yves"],"committee_chairs":[],"committee_members":[],"year":2023,"date_issued":"2023-03-28","date_published":"2023-03-28","updated_at":"2026-07-24T06:33:58Z","subjects":["Computer science"],"languages":["en"],"rights":["Author owns copyright, except where explicitly noted. Please contact the author directly with licensing requests."],"rights_urls":[],"identifier_entries":[]},"links":{"outbound_url":"http://hdl.handle.net/10315/41039","outbound_label":"Handle","outbound_source":"dc:identifier.uri"},"metadata_groups":[{"id":"people","label":"People","entries":[{"key":"dc:contributor.advisor","label":"Advisor","values":["Lesperance, Yves"]},{"key":"dc:creator","label":"Author","values":["Sun, Haolin"]}]},{"id":"academic_context","label":"Academic Context","entries":[{"key":"dc:date.accessioned","label":"Dc Date Accessioned","values":["2023-03-28T21:23:20Z"]},{"key":"dc:date.available","label":"Dc Date Available","values":["2023-03-28T21:23:20Z"]},{"key":"dc:date.issued","label":"Date","values":["2023-03-28"]},{"key":"dc:type","label":"Dc Type","values":["Electronic Thesis or Dissertation"]}]},{"id":"subjects_keywords","label":"Subjects and Keywords","entries":[{"key":"dc:subject","label":"Dc Subject","values":["Computer science"]}]},{"id":"language_rights","label":"Language and Rights","entries":[{"key":"dc:language","label":"Dc Language","values":["en"]},{"key":"dc:rights","label":"Dc Rights","values":["Author owns copyright, except where explicitly noted. Please contact the author directly with licensing requests."]}]},{"id":"identifiers","label":"Identifiers","entries":[{"key":"dc:identifier.uri","label":"Identifier URI","values":["http://hdl.handle.net/10315/41039"]}]},{"id":"additional","label":"Additional Metadata","entries":[{"key":"dc:description.abstract","label":"Abstract","values":["Deep reinforcement learning can solve real-world robot control problems, such as autonomous driving and robotic arm manipulation. In deep reinforcement learning, an agent does not know the problem description and learns the optimal solution through trial-and-error. This method brings two major challenges when solving real-world problems: partial observability and learning efficiency. In this thesis, we address these two challenges and extend previous work. First, we use reward machines to address the problem of partial observability. Then, we focus on finding the existing cutting-edge deep reinforcement learning algorithms and integrating them with reward machines to enhance the learning efficiency. To test the performance of all the algorithms, we proposed a series of different tasks that can be used to mimic real-world robot control problems. Finally, based on the test results, we compare the performance of all the algorithms and analyze their advantages and disadvantages."]},{"key":"dc:title","label":"Title","values":["Exploiting Reward Machines with Deep Reinforcement Learning in Continuous Action Domains"]}]}],"canonical_facts":{"dc:contributor.advisor":["Lesperance, Yves"],"dc:creator":["Sun, Haolin"],"dc:date.accessioned":["2023-03-28T21:23:20Z"],"dc:date.available":["2023-03-28T21:23:20Z"],"dc:date.issued":["2023-03-28"],"dc:description.abstract":["Deep reinforcement learning can solve real-world robot control problems, such as autonomous driving and robotic arm manipulation. In deep reinforcement learning, an agent does not know the problem description and learns the optimal solution through trial-and-error. This method brings two major challenges when solving real-world problems: partial observability and learning efficiency. In this thesis, we address these two challenges and extend previous work. First, we use reward machines to address the problem of partial observability. Then, we focus on finding the existing cutting-edge deep reinforcement learning algorithms and integrating them with reward machines to enhance the learning efficiency. To test the performance of all the algorithms, we proposed a series of different tasks that can be used to mimic real-world robot control problems. Finally, based on the test results, we compare the performance of all the algorithms and analyze their advantages and disadvantages."],"dc:identifier.uri":["http://hdl.handle.net/10315/41039"],"dc:language":["en"],"dc:rights":["Author owns copyright, except where explicitly noted. Please contact the author directly with licensing requests."],"dc:subject":["Computer science"],"dc:title":["Exploiting Reward Machines with Deep Reinforcement Learning in Continuous Action Domains"],"dc:type":["Electronic Thesis or Dissertation"]},"updated_at":"2026-07-24T06:33:58Z"}