{"id":{"repo_id":"uiuc","oai_identifier":"oai:www.ideals.illinois.edu:2142/120420"},"canonical_url":"https://search.dev.ndltd.org/etd/uiuc/oai:www.ideals.illinois.edu:2142/120420","repository":{"repo_id":"uiuc","name":"University of Illinois - Urbana-Champaign","base_url":"https://www.ideals.illinois.edu/oai-pmh"},"display":{"title":"Does the agent say what it sees: an analysis in 3D question answering","abstract":"Submission published under a 24 month embargo labeled 'U of I Access', the embargo will last until 2025-05-01","abstract_html":"Submission published under a 24 month embargo labeled &#x27;U of I Access&#x27;, the embargo will last until 2025-05-01","abstract_has_math":false,"creators":["Zhang, Haomeng"],"institution":"University of Illinois at Urbana-Champaign","degree_name":"M.S.","degree_level":"Thesis","degree_discipline":"Computer Science","degree_department":null,"school":null,"contributors":["Gui, Liangyan"],"advisors":[],"committee_chairs":[],"committee_members":[],"year":2023,"date_issued":"2023-05","date_published":"2023-05","updated_at":"2026-07-22T22:24:57Z","subjects":["Multi-modal Learning","3d Question Answering"],"languages":["en","eng"],"rights":["Copyright 2023 Haomeng Zhang"],"rights_urls":[],"identifier_entries":[]},"links":{"outbound_url":"https://hdl.handle.net/2142/120420","outbound_label":"Handle","outbound_source":"dc:identifier"},"metadata_groups":[{"id":"people","label":"People","entries":[{"key":"dc:contributor","label":"Contributor","values":["Gui, Liangyan"]},{"key":"dc:creator","label":"Author","values":["Zhang, Haomeng"]}]},{"id":"academic_context","label":"Academic Context","entries":[{"key":"dc:date","label":"Dc Date","values":["2023-05","2023-04-27"]},{"key":"dc:type","label":"Dc Type","values":["text","Thesis"]},{"key":"thesis:degree_discipline","label":"Discipline","values":["Computer Science"]},{"key":"thesis:degree_level","label":"Degree Level","values":["Thesis"]},{"key":"thesis:degree_name","label":"Degree Name","values":["M.S."]},{"key":"thesis:institution_name","label":"Thesis Institution Name","values":["University of Illinois at Urbana-Champaign"]}]},{"id":"subjects_keywords","label":"Subjects and Keywords","entries":[{"key":"dc:subject","label":"Dc Subject","values":["Multi-modal Learning","3d Question Answering"]}]},{"id":"language_rights","label":"Language and Rights","entries":[{"key":"dc:language","label":"Dc Language","values":["en","eng"]},{"key":"dc:rights","label":"Dc Rights","values":["Copyright 2023 Haomeng Zhang"]}]},{"id":"identifiers","label":"Identifiers","entries":[{"key":"dc:identifier","label":"Identifier","values":["https://hdl.handle.net/2142/120420"]}]},{"id":"additional","label":"Additional Metadata","entries":[{"key":"dc:description","label":"Description","values":["Submission published under a 24 month embargo labeled 'U of I Access', the embargo will last until 2025-05-01","The student, Haomeng Zhang, accepted the attached license on 2023-04-25 at 16:09.","The student, Haomeng Zhang, submitted this Thesis for approval on 2023-04-25 at 16:21.","This Thesis was approved for publication on 2023-04-27 at 12:56.","DSpace SAF Submission Ingestion Package generated from Vireo submission #19174 on 2023-09-01 at 17:14:55","The emerging research topic of 3D Question and Answering involves complex challenges such as semantic language understanding, 3D scene comprehension, and establishing correspondence between language and target objects. In this thesis, I conduct a comprehensive analysis on the state-of-the-art 3D Question Answering benchmark model, ScanQA, by comparing the effectiveness of various auxiliary tasks, identifying limitations in current evaluation methods, and conducting additional ablation studies on the visual component. The findings and potential future directions for the 3D Question Answering task are discussed to assist researchers in their ongoing investigations."]},{"key":"dc:format","label":"Dc Format","values":["application/pdf"]},{"key":"dc:title","label":"Title","values":["Does the agent say what it sees: an analysis in 3D question answering"]}]}],"canonical_facts":{"dc:contributor":["Gui, Liangyan"],"dc:creator":["Zhang, Haomeng"],"dc:date":["2023-05","2023-04-27"],"dc:description":["Submission published under a 24 month embargo labeled 'U of I Access', the embargo will last until 2025-05-01","The student, Haomeng Zhang, accepted the attached license on 2023-04-25 at 16:09.","The student, Haomeng Zhang, submitted this Thesis for approval on 2023-04-25 at 16:21.","This Thesis was approved for publication on 2023-04-27 at 12:56.","DSpace SAF Submission Ingestion Package generated from Vireo submission #19174 on 2023-09-01 at 17:14:55","The emerging research topic of 3D Question and Answering involves complex challenges such as semantic language understanding, 3D scene comprehension, and establishing correspondence between language and target objects. In this thesis, I conduct a comprehensive analysis on the state-of-the-art 3D Question Answering benchmark model, ScanQA, by comparing the effectiveness of various auxiliary tasks, identifying limitations in current evaluation methods, and conducting additional ablation studies on the visual component. The findings and potential future directions for the 3D Question Answering task are discussed to assist researchers in their ongoing investigations."],"dc:format":["application/pdf"],"dc:identifier":["https://hdl.handle.net/2142/120420"],"dc:language":["en","eng"],"dc:rights":["Copyright 2023 Haomeng Zhang"],"dc:subject":["Multi-modal Learning","3d Question Answering"],"dc:title":["Does the agent say what it sees: an analysis in 3D question answering"],"dc:type":["text","Thesis"],"thesis:degree_discipline":["Computer Science"],"thesis:degree_level":["Thesis"],"thesis:degree_name":["M.S."],"thesis:institution_name":["University of Illinois at Urbana-Champaign"]},"updated_at":"2026-07-22T22:24:57Z"}