{"id":{"repo_id":"njit","oai_identifier":"oai:digitalcommons.njit.edu:theses-1044"},"canonical_url":"https://search.dev.ndltd.org/etd/njit/oai:digitalcommons.njit.edu:theses-1044","repository":{"repo_id":"njit","name":"NJIT","base_url":"https://digitalcommons.njit.edu/do/oai/"},"display":{"title":"Looping predictive method to improve accuracy of a machine learning model","abstract":"The topic of this project is an analysis of drug-related tweets. The goal is to build a Machine Learning Model that can distinguish between tweets that indicate drug abuse and other tweets that also contain the name of a drug but do not describe abuse. Drugs can be illegal, such as heroin, or legal drugs with a potential of abuse, such as painkillers. However, building a good Machine Learning Model requires a large amount of training data. For each training tweet, a human expert has determined whether it indicates drug abuse or not. This is difficult work for humans. In this project a new “Looping Predictive Method” was developed that allows generating large training datasets from a small seed set of tweets by repeatedly adding machine-labeled tweets to the human-labeled tweets. With this method, an accuracy improvement of 15.4% was achieved from an initial set of 1,075 tweets, by expanding the training set to 29,908 tweets.","abstract_html":"The topic of this project is an analysis of drug-related tweets. The goal is to build a Machine Learning Model that can distinguish between tweets that indicate drug abuse and other tweets that also contain the name of a drug but do not describe abuse. Drugs can be illegal, such as heroin, or legal drugs with a potential of abuse, such as painkillers. However, building a good Machine Learning Model requires a large amount of training data. For each training tweet, a human expert has determined whether it indicates drug abuse or not. This is difficult work for humans. In this project a new “Looping Predictive Method” was developed that allows generating large training datasets from a small seed set of tweets by repeatedly adding machine-labeled tweets to the human-labeled tweets. With this method, an accuracy improvement of 15.4% was achieved from an initial set of 1,075 tweets, by expanding the training set to 29,908 tweets.","abstract_has_math":false,"creators":["Pogili, Subramanyam Reddy"],"institution":null,"degree_name":"Master of Science in Computer Science - (M.S.)","degree_level":null,"degree_discipline":"Computer Science","degree_department":null,"school":null,"contributors":["James Geller","Soon Ae Chun","Hai Nhat Phan"],"advisors":[],"committee_chairs":[],"committee_members":[],"year":2017,"date_issued":"2017-12-31T08:00:00Z","date_published":"2017-12-31T08:00:00Z","updated_at":"2026-07-24T03:22:07Z","subjects":["Machine learning","Looping predictive method","Computer Sciences"],"languages":[],"rights":[],"rights_urls":[],"identifier_entries":[]},"links":{"outbound_url":"https://digitalcommons.njit.edu/theses/45","outbound_label":"Repository record","outbound_source":"dc:identifier"},"metadata_groups":[{"id":"people","label":"People","entries":[{"key":"dc:contributor","label":"Contributor","values":["James Geller","Soon Ae Chun","Hai Nhat Phan"]},{"key":"dc:creator","label":"Author","values":["Pogili, Subramanyam Reddy"]}]},{"id":"academic_context","label":"Academic Context","entries":[{"key":"dc:type","label":"Dc Type","values":["Thesis"]},{"key":"thesis:degree_discipline","label":"Discipline","values":["Computer Science"]},{"key":"thesis:degree_name","label":"Degree Name","values":["Master of Science in Computer Science - (M.S.)"]}]},{"id":"subjects_keywords","label":"Subjects and Keywords","entries":[{"key":"dc:subject","label":"Dc Subject","values":["Machine learning","Looping predictive method","Computer Sciences"]}]},{"id":"identifiers","label":"Identifiers","entries":[{"key":"dc:identifier","label":"Identifier","values":["https://digitalcommons.njit.edu/theses/45"]}]},{"id":"additional","label":"Additional Metadata","entries":[{"key":"dc:description.abstract","label":"Abstract","values":["The topic of this project is an analysis of drug-related tweets. The goal is to build a Machine Learning Model that can distinguish between tweets that indicate drug abuse and other tweets that also contain the name of a drug but do not describe abuse. Drugs can be illegal, such as heroin, or legal drugs with a potential of abuse, such as painkillers. However, building a good Machine Learning Model requires a large amount of training data. For each training tweet, a human expert has determined whether it indicates drug abuse or not. This is difficult work for humans. In this project a new “Looping Predictive Method” was developed that allows generating large training datasets from a small seed set of tweets by repeatedly adding machine-labeled tweets to the human-labeled tweets. With this method, an accuracy improvement of 15.4% was achieved from an initial set of 1,075 tweets, by expanding the training set to 29,908 tweets."]},{"key":"dc:title","label":"Title","values":["Looping predictive method to improve accuracy of a machine learning model"]}]}],"canonical_facts":{"dc:contributor":["James Geller","Soon Ae Chun","Hai Nhat Phan"],"dc:creator":["Pogili, Subramanyam Reddy"],"dc:description.abstract":["The topic of this project is an analysis of drug-related tweets. The goal is to build a Machine Learning Model that can distinguish between tweets that indicate drug abuse and other tweets that also contain the name of a drug but do not describe abuse. Drugs can be illegal, such as heroin, or legal drugs with a potential of abuse, such as painkillers. However, building a good Machine Learning Model requires a large amount of training data. For each training tweet, a human expert has determined whether it indicates drug abuse or not. This is difficult work for humans. In this project a new “Looping Predictive Method” was developed that allows generating large training datasets from a small seed set of tweets by repeatedly adding machine-labeled tweets to the human-labeled tweets. With this method, an accuracy improvement of 15.4% was achieved from an initial set of 1,075 tweets, by expanding the training set to 29,908 tweets."],"dc:identifier":["https://digitalcommons.njit.edu/theses/45"],"dc:subject":["Machine learning","Looping predictive method","Computer Sciences"],"dc:title":["Looping predictive method to improve accuracy of a machine learning model"],"dc:type":["Thesis"],"thesis:degree_discipline":["Computer Science"],"thesis:degree_name":["Master of Science in Computer Science - (M.S.)"]},"updated_at":"2026-07-24T03:22:07Z"}