{"id":{"repo_id":"uiuc","oai_identifier":"oai:www.ideals.illinois.edu:2142/115955"},"canonical_url":"https://search.dev.ndltd.org/etd/uiuc/oai:www.ideals.illinois.edu:2142/115955","repository":{"repo_id":"uiuc","name":"University of Illinois - Urbana-Champaign","base_url":"https://www.ideals.illinois.edu/oai-pmh"},"display":{"title":"Microbial named entity recognition using BERT models","abstract":"Submission published under a 24 month embargo labeled 'Closed Access', the embargo will last until 2024-08-01","abstract_html":"Submission published under a 24 month embargo labeled &#x27;Closed Access&#x27;, the embargo will last until 2024-08-01","abstract_has_math":false,"creators":["Rao, Brian K"],"institution":"University of Illinois at Urbana-Champaign","degree_name":"M.S.","degree_level":"Thesis","degree_discipline":"Bioinformatics","degree_department":null,"school":null,"contributors":["Kilicoglu, Halil"],"advisors":[],"committee_chairs":[],"committee_members":[],"year":2022,"date_issued":"2022-08","date_published":"2022-08","updated_at":"2026-07-22T22:24:55Z","subjects":["Bioinformatics"],"languages":["en","eng"],"rights":["Copyright 2022 Brian Rao"],"rights_urls":[],"identifier_entries":[]},"links":{"outbound_url":"https://hdl.handle.net/2142/115955","outbound_label":"Handle","outbound_source":"dc:identifier"},"metadata_groups":[{"id":"people","label":"People","entries":[{"key":"dc:contributor","label":"Contributor","values":["Kilicoglu, Halil"]},{"key":"dc:creator","label":"Author","values":["Rao, Brian K"]}]},{"id":"academic_context","label":"Academic Context","entries":[{"key":"dc:date","label":"Dc Date","values":["2022-08","2022-07-21"]},{"key":"dc:type","label":"Dc Type","values":["text","Thesis"]},{"key":"thesis:degree_discipline","label":"Discipline","values":["Bioinformatics"]},{"key":"thesis:degree_level","label":"Degree Level","values":["Thesis"]},{"key":"thesis:degree_name","label":"Degree Name","values":["M.S."]},{"key":"thesis:institution_name","label":"Thesis Institution Name","values":["University of Illinois at Urbana-Champaign"]}]},{"id":"subjects_keywords","label":"Subjects and Keywords","entries":[{"key":"dc:subject","label":"Dc Subject","values":["Bioinformatics"]}]},{"id":"language_rights","label":"Language and Rights","entries":[{"key":"dc:language","label":"Dc Language","values":["en","eng"]},{"key":"dc:rights","label":"Dc Rights","values":["Copyright 2022 Brian Rao"]}]},{"id":"identifiers","label":"Identifiers","entries":[{"key":"dc:identifier","label":"Identifier","values":["https://hdl.handle.net/2142/115955"]}]},{"id":"additional","label":"Additional Metadata","entries":[{"key":"dc:description","label":"Description","values":["Submission published under a 24 month embargo labeled 'Closed Access', the embargo will last until 2024-08-01","The student, Brian Rao, accepted the attached license on 2022-07-20 at 11:54.","The student, Brian Rao, submitted this Thesis for approval on 2022-07-20 at 13:17.","This Thesis was approved for publication on 2022-07-21 at 08:51.","DSpace SAF Submission Ingestion Package generated from Vireo submission #18392 on 2022-11-16 at 10:56:38","Bacteria are critical subjects of microbiological research that span many rapidly-growing fields of study. The development and widespread application of high-throughput sequencing has led to more microbial data being collected in recent years than ever before. This study investigates the capabilities of the popular Natural Language Processing (NLP) model Bidirectional Encoder Representations from Transformers (BERT) on the relatively understudied text mining domain of microbiology. This is done by fine-tuning a variety of BERT models (BERT, DistilBERT, SciBERT, BioBERT, PubMedBERT) on the Bacteria Biotope 2019 Open Shared Task (BB2019-OST) corpus of annotated microbial research text and evaluating the best performing models on the Named Entity Recognition (NER) task. Following this, an in-depth error analysis was conducted to gain insights into BERT’s entity recognition capabilities. Finally, to investigate performance capabilities further, learning rate and batch size hyperparameters were tuned to increase F1-score. The best BERT model in the comparison was BioBERT, earning an F1-score of 73.82 (±1.04) with default hyperparameters, and 75.35 (±0.62) with tuned hyperparameters. BioBERT had better F1-scores and entity-level statistics, despite PubMedBERT ranking high in biomedical NLP benchmarks. This suggests that the generality of the pretraining corpora of BERT models is particularly important for text mining in the microbial domain."]},{"key":"dc:format","label":"Dc Format","values":["application/pdf"]},{"key":"dc:title","label":"Title","values":["Microbial named entity recognition using BERT models"]}]}],"canonical_facts":{"dc:contributor":["Kilicoglu, Halil"],"dc:creator":["Rao, Brian K"],"dc:date":["2022-08","2022-07-21"],"dc:description":["Submission published under a 24 month embargo labeled 'Closed Access', the embargo will last until 2024-08-01","The student, Brian Rao, accepted the attached license on 2022-07-20 at 11:54.","The student, Brian Rao, submitted this Thesis for approval on 2022-07-20 at 13:17.","This Thesis was approved for publication on 2022-07-21 at 08:51.","DSpace SAF Submission Ingestion Package generated from Vireo submission #18392 on 2022-11-16 at 10:56:38","Bacteria are critical subjects of microbiological research that span many rapidly-growing fields of study. The development and widespread application of high-throughput sequencing has led to more microbial data being collected in recent years than ever before. This study investigates the capabilities of the popular Natural Language Processing (NLP) model Bidirectional Encoder Representations from Transformers (BERT) on the relatively understudied text mining domain of microbiology. This is done by fine-tuning a variety of BERT models (BERT, DistilBERT, SciBERT, BioBERT, PubMedBERT) on the Bacteria Biotope 2019 Open Shared Task (BB2019-OST) corpus of annotated microbial research text and evaluating the best performing models on the Named Entity Recognition (NER) task. Following this, an in-depth error analysis was conducted to gain insights into BERT’s entity recognition capabilities. Finally, to investigate performance capabilities further, learning rate and batch size hyperparameters were tuned to increase F1-score. The best BERT model in the comparison was BioBERT, earning an F1-score of 73.82 (±1.04) with default hyperparameters, and 75.35 (±0.62) with tuned hyperparameters. BioBERT had better F1-scores and entity-level statistics, despite PubMedBERT ranking high in biomedical NLP benchmarks. This suggests that the generality of the pretraining corpora of BERT models is particularly important for text mining in the microbial domain."],"dc:format":["application/pdf"],"dc:identifier":["https://hdl.handle.net/2142/115955"],"dc:language":["en","eng"],"dc:rights":["Copyright 2022 Brian Rao"],"dc:subject":["Bioinformatics"],"dc:title":["Microbial named entity recognition using BERT models"],"dc:type":["text","Thesis"],"thesis:degree_discipline":["Bioinformatics"],"thesis:degree_level":["Thesis"],"thesis:degree_name":["M.S."],"thesis:institution_name":["University of Illinois at Urbana-Champaign"]},"updated_at":"2026-07-22T22:24:55Z"}