{"id":{"repo_id":"washington","oai_identifier":"oai:digital.lib.washington.edu:1773/45931"},"canonical_url":"https://search.dev.ndltd.org/etd/washington/oai:digital.lib.washington.edu:1773/45931","repository":{"repo_id":"washington","name":"University of Washington","base_url":"https://digital.lib.washington.edu/server/oai/request"},"display":{"title":"Learning by Watching and Learning by Doing","abstract":"When we are babies, we learn how to see by watching how the world changes and by interacting with it. Can we use these same signals to train vision models? In this thesis, we outline several works which use these paradigms as a basis for learning algorithms. First, we explore learning by watching in which video data is directly used to learn about the visual world. Second, we tackle multiple challenging tasks in embodied environments in which agents learn by interacting with their surroundings.","abstract_html":"When we are babies, we learn how to see by watching how the world changes and by interacting with it. Can we use these same signals to train vision models? In this thesis, we outline several works which use these paradigms as a basis for learning algorithms. First, we explore learning by watching in which video data is directly used to learn about the visual world. Second, we tackle multiple challenging tasks in embodied environments in which agents learn by interacting with their surroundings.","abstract_has_math":false,"creators":["Gordon, Daniel"],"institution":null,"degree_name":null,"degree_level":null,"degree_discipline":null,"degree_department":null,"school":null,"contributors":[],"advisors":["Farhadi, Ali","Fox, Dieter"],"committee_chairs":[],"committee_members":[],"year":2020,"date_issued":"2020-08-14","date_published":"2020-08-14","updated_at":"2026-07-24T05:58:01Z","subjects":["Computer Vision","Deep Learning","Machine Learning","Computer science","Robotics"],"languages":["en_US"],"rights":["CC BY-SA"],"rights_urls":[],"identifier_entries":[]},"links":{"outbound_url":"http://hdl.handle.net/1773/45931","outbound_label":"Handle","outbound_source":"dc:identifier.uri"},"metadata_groups":[{"id":"people","label":"People","entries":[{"key":"dc:contributor.advisor","label":"Advisor","values":["Farhadi, Ali","Fox, Dieter"]},{"key":"dc:creator","label":"Author","values":["Gordon, Daniel"]}]},{"id":"academic_context","label":"Academic Context","entries":[{"key":"dc:date.accessioned","label":"Dc Date Accessioned","values":["2020-08-14T03:28:36Z"]},{"key":"dc:date.available","label":"Dc Date Available","values":["2020-08-14T03:28:36Z"]},{"key":"dc:date.issued","label":"Date","values":["2020-08-14"]},{"key":"dc:type","label":"Dc Type","values":["Thesis"]}]},{"id":"subjects_keywords","label":"Subjects and Keywords","entries":[{"key":"dc:subject","label":"Dc Subject","values":["Computer Vision","Deep Learning","Machine Learning","Computer science","Robotics"]}]},{"id":"language_rights","label":"Language and Rights","entries":[{"key":"dc:language.iso","label":"Language (ISO)","values":["en_US"]},{"key":"dc:rights","label":"Dc Rights","values":["CC BY-SA"]}]},{"id":"identifiers","label":"Identifiers","entries":[{"key":"dc:identifier.other","label":"Dc Identifier Other","values":["Gordon_washington_0250E_21404.pdf"]},{"key":"dc:identifier.uri","label":"Identifier URI","values":["http://hdl.handle.net/1773/45931"]}]},{"id":"additional","label":"Additional Metadata","entries":[{"key":"dc:description","label":"Description","values":["Thesis (Ph.D.)--University of Washington, 2020"]},{"key":"dc:description.abstract","label":"Abstract","values":["When we are babies, we learn how to see by watching how the world changes and by interacting with it. Can we use these same signals to train vision models? In this thesis, we outline several works which use these paradigms as a basis for learning algorithms. First, we explore learning by watching in which video data is directly used to learn about the visual world. Second, we tackle multiple challenging tasks in embodied environments in which agents learn by interacting with their surroundings."]},{"key":"dc:format.mimetype","label":"Dc Format Mimetype","values":["application/pdf"]},{"key":"dc:title","label":"Title","values":["Learning by Watching and Learning by Doing"]}]}],"canonical_facts":{"dc:contributor.advisor":["Farhadi, Ali","Fox, Dieter"],"dc:creator":["Gordon, Daniel"],"dc:date.accessioned":["2020-08-14T03:28:36Z"],"dc:date.available":["2020-08-14T03:28:36Z"],"dc:date.issued":["2020-08-14"],"dc:description":["Thesis (Ph.D.)--University of Washington, 2020"],"dc:description.abstract":["When we are babies, we learn how to see by watching how the world changes and by interacting with it. Can we use these same signals to train vision models? In this thesis, we outline several works which use these paradigms as a basis for learning algorithms. First, we explore learning by watching in which video data is directly used to learn about the visual world. Second, we tackle multiple challenging tasks in embodied environments in which agents learn by interacting with their surroundings."],"dc:format.mimetype":["application/pdf"],"dc:identifier.other":["Gordon_washington_0250E_21404.pdf"],"dc:identifier.uri":["http://hdl.handle.net/1773/45931"],"dc:language.iso":["en_US"],"dc:rights":["CC BY-SA"],"dc:subject":["Computer Vision","Deep Learning","Machine Learning","Computer science","Robotics"],"dc:title":["Learning by Watching and Learning by Doing"],"dc:type":["Thesis"]},"updated_at":"2026-07-24T05:58:01Z"}