{"id":{"repo_id":"mit","oai_identifier":"oai:dspace.mit.edu:1721.1/84878"},"canonical_url":"https://search.dev.ndltd.org/etd/mit/oai:dspace.mit.edu:1721.1/84878","repository":{"repo_id":"mit","name":"MIT","base_url":"https://dspace.mit.edu/oai/request"},"display":{"title":"Word sense disambiguation in clinical text","abstract":"Lexical ambiguity, the ambiguity arising from a string with multiple meanings, is pervasive in language of all domains. Word sense disambiguation (WSD) and word sense induction (WSI) are the tasks of resolving this ambiguity. Applications in the clinical and biomedical domain focus on the potential disambiguation has for information extraction. Most approaches to the problem are unsupervised or semi-supervised because of the high cost of obtaining enough annotated data for supervised learning. In this thesis we compare the application of a semi-supervised general domain state of the art WSI method to clinical text to the best known knowledge-based unsupervised methods in the clinical domain. We also explore making improvements to the general domain method, which is based on topic modeling, by adding features that incorporate syntax and information from knowledge bases, and investigate ways to mitigate the need for annotated data.","abstract_html":"Lexical ambiguity, the ambiguity arising from a string with multiple meanings, is pervasive in language of all domains. Word sense disambiguation (WSD) and word sense induction (WSI) are the tasks of resolving this ambiguity. Applications in the clinical and biomedical domain focus on the potential disambiguation has for information extraction. Most approaches to the problem are unsupervised or semi-supervised because of the high cost of obtaining enough annotated data for supervised learning. In this thesis we compare the application of a semi-supervised general domain state of the art WSI method to clinical text to the best known knowledge-based unsupervised methods in the clinical domain. We also explore making improvements to the general domain method, which is based on topic modeling, by adding features that incorporate syntax and information from knowledge bases, and investigate ways to mitigate the need for annotated data.","abstract_has_math":false,"creators":["Chasin, Rachel (Rachel G.)"],"institution":"Massachusetts Institute of Technology","degree_name":null,"degree_level":null,"degree_discipline":null,"degree_department":"Massachusetts Institute of Technology. Department of Electrical Engineering and Computer Science.","school":null,"contributors":[],"advisors":["Peter Szolovits."],"committee_chairs":[],"committee_members":[],"year":2013,"date_issued":"2013","date_published":"2013","updated_at":"2026-07-22T22:21:30Z","subjects":["Electrical Engineering and Computer Science."],"languages":["eng"],"rights":["M.I.T. theses are protected by copyright. They may be viewed from this source for any purpose, but reproduction or distribution in any format is prohibited without written permission. See provided URL for inquiries about permission."],"rights_urls":["http://dspace.mit.edu/handle/1721.1/7582"],"identifier_entries":[]},"links":{"outbound_url":"http://hdl.handle.net/1721.1/84878","outbound_label":"Handle","outbound_source":"dc:identifier.uri"},"metadata_groups":[{"id":"people","label":"People","entries":[{"key":"dc:contributor.advisor","label":"Advisor","values":["Peter Szolovits."]},{"key":"dc:contributor.department","label":"Department","values":["Massachusetts Institute of Technology. Department of Electrical Engineering and Computer Science."]},{"key":"dc:contributor.other","label":"Dc Contributor Other","values":["Massachusetts Institute of Technology. Department of Electrical Engineering and Computer Science."]},{"key":"dc:creator","label":"Author","values":["Chasin, Rachel (Rachel G.)"]}]},{"id":"academic_context","label":"Academic Context","entries":[{"key":"dc:date.accessioned","label":"Dc Date Accessioned","values":["2014-02-10T16:57:36Z"]},{"key":"dc:date.available","label":"Dc Date Available","values":["2014-02-10T16:57:36Z"]},{"key":"dc:date.issued","label":"Date","values":["2013"]},{"key":"dc:publisher","label":"Institution","values":["Massachusetts Institute of Technology"]},{"key":"dc:type","label":"Dc Type","values":["Thesis"]}]},{"id":"subjects_keywords","label":"Subjects and Keywords","entries":[{"key":"dc:subject","label":"Dc Subject","values":["Electrical Engineering and Computer Science."]}]},{"id":"language_rights","label":"Language and Rights","entries":[{"key":"dc:language.iso","label":"Language (ISO)","values":["eng"]},{"key":"dc:rights","label":"Dc Rights","values":["M.I.T. theses are protected by copyright. They may be viewed from this source for any purpose, but reproduction or distribution in any format is prohibited without written permission. See provided URL for inquiries about permission."]},{"key":"dc:rights.uri","label":"Rights URI","values":["http://dspace.mit.edu/handle/1721.1/7582"]}]},{"id":"identifiers","label":"Identifiers","entries":[{"key":"dc:identifier.uri","label":"Identifier URI","values":["http://hdl.handle.net/1721.1/84878"]}]},{"id":"additional","label":"Additional Metadata","entries":[{"key":"dc:description","label":"Description","values":["Thesis (M. Eng.)--Massachusetts Institute of Technology, Department of Electrical Engineering and Computer Science, 2013.","Cataloged from PDF version of thesis.","Includes bibliographical references (pages 57-59)."]},{"key":"dc:description.abstract","label":"Abstract","values":["Lexical ambiguity, the ambiguity arising from a string with multiple meanings, is pervasive in language of all domains. Word sense disambiguation (WSD) and word sense induction (WSI) are the tasks of resolving this ambiguity. Applications in the clinical and biomedical domain focus on the potential disambiguation has for information extraction. Most approaches to the problem are unsupervised or semi-supervised because of the high cost of obtaining enough annotated data for supervised learning. In this thesis we compare the application of a semi-supervised general domain state of the art WSI method to clinical text to the best known knowledge-based unsupervised methods in the clinical domain. We also explore making improvements to the general domain method, which is based on topic modeling, by adding features that incorporate syntax and information from knowledge bases, and investigate ways to mitigate the need for annotated data."]},{"key":"dc:description.degree","label":"Dc Description Degree","values":["M.Eng."]},{"key":"dc:title","label":"Title","values":["Word sense disambiguation in clinical text"]}]}],"canonical_facts":{"dc:contributor.advisor":["Peter Szolovits."],"dc:contributor.department":["Massachusetts Institute of Technology. Department of Electrical Engineering and Computer Science."],"dc:contributor.other":["Massachusetts Institute of Technology. Department of Electrical Engineering and Computer Science."],"dc:creator":["Chasin, Rachel (Rachel G.)"],"dc:date.accessioned":["2014-02-10T16:57:36Z"],"dc:date.available":["2014-02-10T16:57:36Z"],"dc:date.issued":["2013"],"dc:description":["Thesis (M. Eng.)--Massachusetts Institute of Technology, Department of Electrical Engineering and Computer Science, 2013.","Cataloged from PDF version of thesis.","Includes bibliographical references (pages 57-59)."],"dc:description.abstract":["Lexical ambiguity, the ambiguity arising from a string with multiple meanings, is pervasive in language of all domains. Word sense disambiguation (WSD) and word sense induction (WSI) are the tasks of resolving this ambiguity. Applications in the clinical and biomedical domain focus on the potential disambiguation has for information extraction. Most approaches to the problem are unsupervised or semi-supervised because of the high cost of obtaining enough annotated data for supervised learning. In this thesis we compare the application of a semi-supervised general domain state of the art WSI method to clinical text to the best known knowledge-based unsupervised methods in the clinical domain. We also explore making improvements to the general domain method, which is based on topic modeling, by adding features that incorporate syntax and information from knowledge bases, and investigate ways to mitigate the need for annotated data."],"dc:description.degree":["M.Eng."],"dc:identifier.uri":["http://hdl.handle.net/1721.1/84878"],"dc:language.iso":["eng"],"dc:publisher":["Massachusetts Institute of Technology"],"dc:rights":["M.I.T. theses are protected by copyright. They may be viewed from this source for any purpose, but reproduction or distribution in any format is prohibited without written permission. See provided URL for inquiries about permission."],"dc:rights.uri":["http://dspace.mit.edu/handle/1721.1/7582"],"dc:subject":["Electrical Engineering and Computer Science."],"dc:title":["Word sense disambiguation in clinical text"],"dc:type":["Thesis"]},"updated_at":"2026-07-22T22:21:30Z"}