{"id":{"repo_id":"uiuc","oai_identifier":"oai:www.ideals.illinois.edu:2142/122213"},"canonical_url":"https://search.dev.ndltd.org/etd/uiuc/oai:www.ideals.illinois.edu:2142/122213","repository":{"repo_id":"uiuc","name":"University of Illinois - Urbana-Champaign","base_url":"https://www.ideals.illinois.edu/oai-pmh"},"display":{"title":"Granular text classification for biomedical natural language processing","abstract":"Submission published under a 24 month embargo labeled 'Closed Access', the embargo will last until 2025-12-01","abstract_html":"Submission published under a 24 month embargo labeled &#x27;Closed Access&#x27;, the embargo will last until 2025-12-01","abstract_has_math":false,"creators":["Abdar, Omid"],"institution":"University of Illinois at Urbana-Champaign","degree_name":"Ph.D.","degree_level":"Dissertation","degree_discipline":"Linguistics","degree_department":null,"school":null,"contributors":["Schwartz, Lane","Stevens, Jon","Markee, Numa","Sadler, Randall","Ionin, Tania"],"advisors":[],"committee_chairs":[],"committee_members":[],"year":2023,"date_issued":"2023-12","date_published":"2023-12","updated_at":"2026-07-22T22:25:00Z","subjects":["Natural Language Processing","Nlp","Biomedical Natural Language Processing","Text Classification","Transformers","Granularity"],"languages":["en","eng"],"rights":["Copyright 2023 Omid Abdar"],"rights_urls":[],"identifier_entries":[]},"links":{"outbound_url":"https://hdl.handle.net/2142/122213","outbound_label":"Handle","outbound_source":"dc:identifier"},"metadata_groups":[{"id":"people","label":"People","entries":[{"key":"dc:contributor","label":"Contributor","values":["Schwartz, Lane","Stevens, Jon","Markee, Numa","Sadler, Randall","Ionin, Tania"]},{"key":"dc:creator","label":"Author","values":["Abdar, Omid"]}]},{"id":"academic_context","label":"Academic Context","entries":[{"key":"dc:date","label":"Dc Date","values":["2023-12","2023-11-15"]},{"key":"dc:type","label":"Dc Type","values":["text"]},{"key":"thesis:degree_discipline","label":"Discipline","values":["Linguistics"]},{"key":"thesis:degree_level","label":"Degree Level","values":["Dissertation"]},{"key":"thesis:degree_name","label":"Degree Name","values":["Ph.D."]},{"key":"thesis:institution_name","label":"Thesis Institution Name","values":["University of Illinois at Urbana-Champaign"]}]},{"id":"subjects_keywords","label":"Subjects and Keywords","entries":[{"key":"dc:subject","label":"Dc Subject","values":["Natural Language Processing","Nlp","Biomedical Natural Language Processing","Text Classification","Transformers","Granularity"]}]},{"id":"language_rights","label":"Language and Rights","entries":[{"key":"dc:language","label":"Dc Language","values":["en","eng"]},{"key":"dc:rights","label":"Dc Rights","values":["Copyright 2023 Omid Abdar"]}]},{"id":"identifiers","label":"Identifiers","entries":[{"key":"dc:identifier","label":"Identifier","values":["https://hdl.handle.net/2142/122213"]}]},{"id":"additional","label":"Additional Metadata","entries":[{"key":"dc:description","label":"Description","values":["Submission published under a 24 month embargo labeled 'Closed Access', the embargo will last until 2025-12-01","The student, Omid Abdar, accepted the attached license on 2023-11-09 at 09:33.","The student, Omid Abdar, submitted this Dissertation for approval on 2023-11-09 at 09:48.","This Dissertation was approved for publication on 2023-11-15 at 08:07.","DSpace SAF Submission Ingestion Package generated from Vireo submission #19899 on 2024-03-01 at 13:48:04","Text classification is a classic NLP problem with numerous applications and use cases in sentiment analysis, spam detection, document organization, and information retrieval systems. While text classification techniques can be applied to documents of different lengths, supervised machine learning approaches for text classification require labeled documents similar in length to those that will be used at inference. This dissertation examines how labeled documents at higher levels of linguistic granularity (i.e., longer documents) may be synthesized to develop text classifiers at a lower level of linguistic granularity (i.e., shorter text). More specifically, we focus on Biomedical Natural Language Processing as our target domain and address (1) how well document-level classifiers perform for sentence-level classification; (2) how document-level labeled data may be synthesized to perform sentence-level text classification; and (3) how the performance of synthesized approaches compares against benchmark performances using labeled data. We present feature contribution analysis experiments in Naïve Bayes classifiers as well as self-attention experiments in BERT, and report sentence classification results that beat baseline performance for both model types. Finally, we present an extensive qualitative error analysis to identify major error trends in our results and discuss the significance of each error category with respect to our primary research questions."]},{"key":"dc:format","label":"Dc Format","values":["application/pdf"]},{"key":"dc:title","label":"Title","values":["Granular text classification for biomedical natural language processing"]}]}],"canonical_facts":{"dc:contributor":["Schwartz, Lane","Stevens, Jon","Markee, Numa","Sadler, Randall","Ionin, Tania"],"dc:creator":["Abdar, Omid"],"dc:date":["2023-12","2023-11-15"],"dc:description":["Submission published under a 24 month embargo labeled 'Closed Access', the embargo will last until 2025-12-01","The student, Omid Abdar, accepted the attached license on 2023-11-09 at 09:33.","The student, Omid Abdar, submitted this Dissertation for approval on 2023-11-09 at 09:48.","This Dissertation was approved for publication on 2023-11-15 at 08:07.","DSpace SAF Submission Ingestion Package generated from Vireo submission #19899 on 2024-03-01 at 13:48:04","Text classification is a classic NLP problem with numerous applications and use cases in sentiment analysis, spam detection, document organization, and information retrieval systems. While text classification techniques can be applied to documents of different lengths, supervised machine learning approaches for text classification require labeled documents similar in length to those that will be used at inference. This dissertation examines how labeled documents at higher levels of linguistic granularity (i.e., longer documents) may be synthesized to develop text classifiers at a lower level of linguistic granularity (i.e., shorter text). More specifically, we focus on Biomedical Natural Language Processing as our target domain and address (1) how well document-level classifiers perform for sentence-level classification; (2) how document-level labeled data may be synthesized to perform sentence-level text classification; and (3) how the performance of synthesized approaches compares against benchmark performances using labeled data. We present feature contribution analysis experiments in Naïve Bayes classifiers as well as self-attention experiments in BERT, and report sentence classification results that beat baseline performance for both model types. Finally, we present an extensive qualitative error analysis to identify major error trends in our results and discuss the significance of each error category with respect to our primary research questions."],"dc:format":["application/pdf"],"dc:identifier":["https://hdl.handle.net/2142/122213"],"dc:language":["en","eng"],"dc:rights":["Copyright 2023 Omid Abdar"],"dc:subject":["Natural Language Processing","Nlp","Biomedical Natural Language Processing","Text Classification","Transformers","Granularity"],"dc:title":["Granular text classification for biomedical natural language processing"],"dc:type":["text"],"thesis:degree_discipline":["Linguistics"],"thesis:degree_level":["Dissertation"],"thesis:degree_name":["Ph.D."],"thesis:institution_name":["University of Illinois at Urbana-Champaign"]},"updated_at":"2026-07-22T22:25:00Z"}