{"id":{"repo_id":"mit","oai_identifier":"oai:dspace.mit.edu:1721.1/29685"},"canonical_url":"https://search.dev.ndltd.org/etd/mit/oai:dspace.mit.edu:1721.1/29685","repository":{"repo_id":"mit","name":"MIT","base_url":"https://dspace.mit.edu/oai/request"},"display":{"title":"An implementation of face-to-face grounding in an embodied conversational agent","abstract":"When people have a face-to-face conversation, they don't just spout information blindly-they work to make sure that both participants understand what has been said. This process of ensuring that what has been said is added to the common ground of the conversation is called grounding. This thesis explores recent research into the verbal and nonverbal means for grounding, and presents an implementation of a face-to-face grounding system in an embodied conversational agent that is based on a model of grounding extracted from the research. This is the first such agent that supports nonverbal grounding, and so this thesis represents both a proof of concept and a guide for future work in this area, showing that it is possible to build a dialogue system that implements face-to-face grounding between a human and an agent based on an empirically-derived model. Additionally, this thesis describes a vision system, based on a stereo-camera head-pose tracker and using a recently proposed method for head-nod detection, that can robustly and accurately identify head nods and gaze state.","abstract_html":"When people have a face-to-face conversation, they don&#x27;t just spout information blindly-they work to make sure that both participants understand what has been said. This process of ensuring that what has been said is added to the common ground of the conversation is called grounding. This thesis explores recent research into the verbal and nonverbal means for grounding, and presents an implementation of a face-to-face grounding system in an embodied conversational agent that is based on a model of grounding extracted from the research. This is the first such agent that supports nonverbal grounding, and so this thesis represents both a proof of concept and a guide for future work in this area, showing that it is possible to build a dialogue system that implements face-to-face grounding between a human and an agent based on an empirically-derived model. Additionally, this thesis describes a vision system, based on a stereo-camera head-pose tracker and using a recently proposed method for head-nod detection, that can robustly and accurately identify head nods and gaze state.","abstract_has_math":false,"creators":["Reinstein, Gabriel A. (Gabriel Alexander), 1980-"],"institution":"Massachusetts Institute of Technology","degree_name":null,"degree_level":null,"degree_discipline":null,"degree_department":"Massachusetts Institute of Technology. Dept. of Electrical Engineering and Computer Science.","school":null,"contributors":[],"advisors":["Justine Cassell."],"committee_chairs":[],"committee_members":[],"year":2003,"date_issued":"2003","date_published":"2003","updated_at":"2026-07-22T22:21:16Z","subjects":["Electrical Engineering and Computer Science."],"languages":["eng"],"rights":["M.I.T. theses are protected by copyright. They may be viewed from this source for any purpose, but reproduction or distribution in any format is prohibited without written permission. See provided URL for inquiries about permission."],"rights_urls":["http://dspace.mit.edu/handle/1721.1/7582"],"identifier_entries":[]},"links":{"outbound_url":"http://hdl.handle.net/1721.1/29685","outbound_label":"Handle","outbound_source":"dc:identifier.uri"},"metadata_groups":[{"id":"people","label":"People","entries":[{"key":"dc:contributor.advisor","label":"Advisor","values":["Justine Cassell."]},{"key":"dc:contributor.department","label":"Department","values":["Massachusetts Institute of Technology. Dept. of Electrical Engineering and Computer Science."]},{"key":"dc:contributor.other","label":"Dc Contributor Other","values":["Massachusetts Institute of Technology. Dept. of Electrical Engineering and Computer Science."]},{"key":"dc:creator","label":"Author","values":["Reinstein, Gabriel A. (Gabriel Alexander), 1980-"]}]},{"id":"academic_context","label":"Academic Context","entries":[{"key":"dc:date.accessioned","label":"Dc Date Accessioned","values":["2006-03-24T16:15:04Z"]},{"key":"dc:date.available","label":"Dc Date Available","values":["2006-03-24T16:15:04Z"]},{"key":"dc:date.issued","label":"Date","values":["2003"]},{"key":"dc:publisher","label":"Institution","values":["Massachusetts Institute of Technology"]},{"key":"dc:type","label":"Dc Type","values":["Thesis"]}]},{"id":"subjects_keywords","label":"Subjects and Keywords","entries":[{"key":"dc:subject","label":"Dc Subject","values":["Electrical Engineering and Computer Science."]}]},{"id":"language_rights","label":"Language and Rights","entries":[{"key":"dc:language.iso","label":"Language (ISO)","values":["eng"]},{"key":"dc:rights","label":"Dc Rights","values":["M.I.T. theses are protected by copyright. They may be viewed from this source for any purpose, but reproduction or distribution in any format is prohibited without written permission. See provided URL for inquiries about permission."]},{"key":"dc:rights.uri","label":"Rights URI","values":["http://dspace.mit.edu/handle/1721.1/7582"]}]},{"id":"identifiers","label":"Identifiers","entries":[{"key":"dc:identifier.uri","label":"Identifier URI","values":["http://hdl.handle.net/1721.1/29685"]}]},{"id":"additional","label":"Additional Metadata","entries":[{"key":"dc:description","label":"Description","values":["Thesis (M.Eng.)--Massachusetts Institute of Technology, Dept. of Electrical Engineering and Computer Science, 2003.","Includes bibliographical references (leaves 53-55)."]},{"key":"dc:description.abstract","label":"Abstract","values":["When people have a face-to-face conversation, they don't just spout information blindly-they work to make sure that both participants understand what has been said. This process of ensuring that what has been said is added to the common ground of the conversation is called grounding. This thesis explores recent research into the verbal and nonverbal means for grounding, and presents an implementation of a face-to-face grounding system in an embodied conversational agent that is based on a model of grounding extracted from the research. This is the first such agent that supports nonverbal grounding, and so this thesis represents both a proof of concept and a guide for future work in this area, showing that it is possible to build a dialogue system that implements face-to-face grounding between a human and an agent based on an empirically-derived model. Additionally, this thesis describes a vision system, based on a stereo-camera head-pose tracker and using a recently proposed method for head-nod detection, that can robustly and accurately identify head nods and gaze state."]},{"key":"dc:description.degree","label":"Dc Description Degree","values":["M.Eng."]},{"key":"dc:format.mimetype","label":"Dc Format Mimetype","values":["application/pdf"]},{"key":"dc:title","label":"Title","values":["An implementation of face-to-face grounding in an embodied conversational agent"]}]}],"canonical_facts":{"dc:contributor.advisor":["Justine Cassell."],"dc:contributor.department":["Massachusetts Institute of Technology. Dept. of Electrical Engineering and Computer Science."],"dc:contributor.other":["Massachusetts Institute of Technology. Dept. of Electrical Engineering and Computer Science."],"dc:creator":["Reinstein, Gabriel A. (Gabriel Alexander), 1980-"],"dc:date.accessioned":["2006-03-24T16:15:04Z"],"dc:date.available":["2006-03-24T16:15:04Z"],"dc:date.issued":["2003"],"dc:description":["Thesis (M.Eng.)--Massachusetts Institute of Technology, Dept. of Electrical Engineering and Computer Science, 2003.","Includes bibliographical references (leaves 53-55)."],"dc:description.abstract":["When people have a face-to-face conversation, they don't just spout information blindly-they work to make sure that both participants understand what has been said. This process of ensuring that what has been said is added to the common ground of the conversation is called grounding. This thesis explores recent research into the verbal and nonverbal means for grounding, and presents an implementation of a face-to-face grounding system in an embodied conversational agent that is based on a model of grounding extracted from the research. This is the first such agent that supports nonverbal grounding, and so this thesis represents both a proof of concept and a guide for future work in this area, showing that it is possible to build a dialogue system that implements face-to-face grounding between a human and an agent based on an empirically-derived model. Additionally, this thesis describes a vision system, based on a stereo-camera head-pose tracker and using a recently proposed method for head-nod detection, that can robustly and accurately identify head nods and gaze state."],"dc:description.degree":["M.Eng."],"dc:format.mimetype":["application/pdf"],"dc:identifier.uri":["http://hdl.handle.net/1721.1/29685"],"dc:language.iso":["eng"],"dc:publisher":["Massachusetts Institute of Technology"],"dc:rights":["M.I.T. theses are protected by copyright. They may be viewed from this source for any purpose, but reproduction or distribution in any format is prohibited without written permission. See provided URL for inquiries about permission."],"dc:rights.uri":["http://dspace.mit.edu/handle/1721.1/7582"],"dc:subject":["Electrical Engineering and Computer Science."],"dc:title":["An implementation of face-to-face grounding in an embodied conversational agent"],"dc:type":["Thesis"]},"updated_at":"2026-07-22T22:21:16Z"}