{"id":{"repo_id":"ksu","oai_identifier":"oai:krex.k-state.edu:2097/44640"},"canonical_url":"https://search.dev.ndltd.org/etd/ksu/oai:krex.k-state.edu:2097/44640","repository":{"repo_id":"ksu","name":"Kansas State University","base_url":"https://krex.k-state.edu/server/oai/request"},"display":{"title":"Leveraging a natural language processing approach towards a more informed vulnerability documentation process","abstract":"Cybersecurity vulnerabilities are an ever-increasing threat to the current cybersecurity landscape. It has been previously suggested that Twitter is a robust data source for gathering Cyber Threat Intelligence data. This includes cyber vulnerabilities which can be retrieved via their Common Vulnerabilities and Exposures (CVE) identifier. However, the culture of post-disclosure vulnerability discussion is changing to sometimes include a ”nickname”, or a short name utilized instead of the CVE identifier. This trend poses a significant challenge to the retrieval of CVE-relevant information as not all text includes the CVE identifier. To address this challenge, a system was designed by utilizing an off-the-shelf machine learning model to link tweets that do not explicitly mention a CVE Identifier to their corresponding CVE. The system was tested utilizing several datasets and metrics to determine parameters required to obtain satisfactory performance with regards to retrieved information. The results show that machine learning makes it possible to retrieve relevant information corresponding to a specific CVE in the absence of the CVE identifier.","abstract_html":"Cybersecurity vulnerabilities are an ever-increasing threat to the current cybersecurity landscape. It has been previously suggested that Twitter is a robust data source for gathering Cyber Threat Intelligence data. This includes cyber vulnerabilities which can be retrieved via their Common Vulnerabilities and Exposures (CVE) identifier. However, the culture of post-disclosure vulnerability discussion is changing to sometimes include a ”nickname”, or a short name utilized instead of the CVE identifier. This trend poses a significant challenge to the retrieval of CVE-relevant information as not all text includes the CVE identifier. To address this challenge, a system was designed by utilizing an off-the-shelf machine learning model to link tweets that do not explicitly mention a CVE Identifier to their corresponding CVE. The system was tested utilizing several datasets and metrics to determine parameters required to obtain satisfactory performance with regards to retrieved information. The results show that machine learning makes it possible to retrieve relevant information corresponding to a specific CVE in the absence of the CVE identifier.","abstract_has_math":false,"creators":["Anshutz, BreAnn Marie"],"institution":null,"degree_name":null,"degree_level":null,"degree_discipline":null,"degree_department":null,"school":null,"contributors":[],"advisors":[],"committee_chairs":[],"committee_members":[],"year":2024,"date_issued":"2024","date_published":"2024","updated_at":"2026-07-27T20:01:31Z","subjects":["Cybersecurity","Cyber Threat Intel","Twitter","Natural language processing"],"languages":["en_US"],"rights":[],"rights_urls":[],"identifier_entries":[]},"links":{"outbound_url":"https://hdl.handle.net/2097/44640","outbound_label":"Handle","outbound_source":"dc:identifier.uri"},"metadata_groups":[{"id":"people","label":"People","entries":[{"key":"dc:creator","label":"Author","values":["Anshutz, BreAnn Marie"]}]},{"id":"academic_context","label":"Academic Context","entries":[{"key":"dc:date.accessioned","label":"Dc Date Accessioned","values":["2024-10-22T18:55:13Z"]},{"key":"dc:date.available","label":"Dc Date Available","values":["2024-10-22T18:55:13Z"]},{"key":"dc:date.issued","label":"Date","values":["2024"]},{"key":"dc:type","label":"Dc Type","values":["Thesis"]}]},{"id":"subjects_keywords","label":"Subjects and Keywords","entries":[{"key":"dc:subject","label":"Dc Subject","values":["Cybersecurity","Cyber Threat Intel","Twitter","Natural language processing"]}]},{"id":"language_rights","label":"Language and Rights","entries":[{"key":"dc:language.iso","label":"Language (ISO)","values":["en_US"]}]},{"id":"identifiers","label":"Identifiers","entries":[{"key":"dc:identifier.uri","label":"Identifier URI","values":["https://hdl.handle.net/2097/44640"]}]},{"id":"additional","label":"Additional Metadata","entries":[{"key":"dc:description.abstract","label":"Abstract","values":["Cybersecurity vulnerabilities are an ever-increasing threat to the current cybersecurity landscape. It has been previously suggested that Twitter is a robust data source for gathering Cyber Threat Intelligence data. This includes cyber vulnerabilities which can be retrieved via their Common Vulnerabilities and Exposures (CVE) identifier. However, the culture of post-disclosure vulnerability discussion is changing to sometimes include a ”nickname”, or a short name utilized instead of the CVE identifier. This trend poses a significant challenge to the retrieval of CVE-relevant information as not all text includes the CVE identifier. To address this challenge, a system was designed by utilizing an off-the-shelf machine learning model to link tweets that do not explicitly mention a CVE Identifier to their corresponding CVE. The system was tested utilizing several datasets and metrics to determine parameters required to obtain satisfactory performance with regards to retrieved information. The results show that machine learning makes it possible to retrieve relevant information corresponding to a specific CVE in the absence of the CVE identifier."]},{"key":"dc:description.degree","label":"Dc Description Degree","values":["Master of Science"]},{"key":"dc:title","label":"Title","values":["Leveraging a natural language processing approach towards a more informed vulnerability documentation process"]}]}],"canonical_facts":{"dc:creator":["Anshutz, BreAnn Marie"],"dc:date.accessioned":["2024-10-22T18:55:13Z"],"dc:date.available":["2024-10-22T18:55:13Z"],"dc:date.issued":["2024"],"dc:description.abstract":["Cybersecurity vulnerabilities are an ever-increasing threat to the current cybersecurity landscape. It has been previously suggested that Twitter is a robust data source for gathering Cyber Threat Intelligence data. This includes cyber vulnerabilities which can be retrieved via their Common Vulnerabilities and Exposures (CVE) identifier. However, the culture of post-disclosure vulnerability discussion is changing to sometimes include a ”nickname”, or a short name utilized instead of the CVE identifier. This trend poses a significant challenge to the retrieval of CVE-relevant information as not all text includes the CVE identifier. To address this challenge, a system was designed by utilizing an off-the-shelf machine learning model to link tweets that do not explicitly mention a CVE Identifier to their corresponding CVE. The system was tested utilizing several datasets and metrics to determine parameters required to obtain satisfactory performance with regards to retrieved information. The results show that machine learning makes it possible to retrieve relevant information corresponding to a specific CVE in the absence of the CVE identifier."],"dc:description.degree":["Master of Science"],"dc:identifier.uri":["https://hdl.handle.net/2097/44640"],"dc:language.iso":["en_US"],"dc:subject":["Cybersecurity","Cyber Threat Intel","Twitter","Natural language processing"],"dc:title":["Leveraging a natural language processing approach towards a more informed vulnerability documentation process"],"dc:type":["Thesis"]},"updated_at":"2026-07-27T20:01:31Z"}