{"id":{"repo_id":"njit","oai_identifier":"oai:digitalcommons.njit.edu:theses-1024"},"canonical_url":"https://search.dev.ndltd.org/etd/njit/oai:digitalcommons.njit.edu:theses-1024","repository":{"repo_id":"njit","name":"NJIT","base_url":"https://digitalcommons.njit.edu/do/oai/"},"display":{"title":"Decision tree rule-based feature selection for imbalanced data","abstract":"A class imbalance problem appears in many real world applications, e.g., fault diagnosis, text categorization and fraud detection. When dealing with an imbalanced dataset, feature selection becomes an important issue. To address it, this work proposes a feature selection method that is based on a decision tree rule and weighted Gini index. The effectiveness of the proposed methods is verified by classifying a dataset from Santander Bank and two datasets from UCI machine learning repository. The results show that our methods can achieve higher Area Under the Curve (AUC) and F-measure. We also compare them with filter-based feature selection approaches, i.e., Chi-Square and F-statistic. The results show that they outperform them but need slightly more computational efforts.","abstract_html":"A class imbalance problem appears in many real world applications, e.g., fault diagnosis, text categorization and fraud detection. When dealing with an imbalanced dataset, feature selection becomes an important issue. To address it, this work proposes a feature selection method that is based on a decision tree rule and weighted Gini index. The effectiveness of the proposed methods is verified by classifying a dataset from Santander Bank and two datasets from UCI machine learning repository. The results show that our methods can achieve higher Area Under the Curve (AUC) and F-measure. We also compare them with filter-based feature selection approaches, i.e., Chi-Square and F-statistic. The results show that they outperform them but need slightly more computational efforts.","abstract_has_math":false,"creators":["Liu, Haoyue"],"institution":null,"degree_name":"Master of Science in Computer Engineering - (M.S.)","degree_level":null,"degree_discipline":"Electrical and Computer Engineering","degree_department":null,"school":null,"contributors":["MengChu Zhou","Osvaldo Simeone","Yun Q. Shi"],"advisors":[],"committee_chairs":[],"committee_members":[],"year":2017,"date_issued":"2017-05-31T07:00:00Z","date_published":"2017-05-31T07:00:00Z","updated_at":"2026-07-24T03:22:01Z","subjects":["Imbalanced data","Decision tree","Computer Engineering"],"languages":[],"rights":[],"rights_urls":[],"identifier_entries":[]},"links":{"outbound_url":"https://digitalcommons.njit.edu/theses/25","outbound_label":"Repository record","outbound_source":"dc:identifier"},"metadata_groups":[{"id":"people","label":"People","entries":[{"key":"dc:contributor","label":"Contributor","values":["MengChu Zhou","Osvaldo Simeone","Yun Q. Shi"]},{"key":"dc:creator","label":"Author","values":["Liu, Haoyue"]}]},{"id":"academic_context","label":"Academic Context","entries":[{"key":"dc:type","label":"Dc Type","values":["Thesis"]},{"key":"thesis:degree_discipline","label":"Discipline","values":["Electrical and Computer Engineering"]},{"key":"thesis:degree_name","label":"Degree Name","values":["Master of Science in Computer Engineering - (M.S.)"]}]},{"id":"subjects_keywords","label":"Subjects and Keywords","entries":[{"key":"dc:subject","label":"Dc Subject","values":["Imbalanced data","Decision tree","Computer Engineering"]}]},{"id":"identifiers","label":"Identifiers","entries":[{"key":"dc:identifier","label":"Identifier","values":["https://digitalcommons.njit.edu/theses/25"]}]},{"id":"additional","label":"Additional Metadata","entries":[{"key":"dc:description.abstract","label":"Abstract","values":["A class imbalance problem appears in many real world applications, e.g., fault diagnosis, text categorization and fraud detection. When dealing with an imbalanced dataset, feature selection becomes an important issue. To address it, this work proposes a feature selection method that is based on a decision tree rule and weighted Gini index. The effectiveness of the proposed methods is verified by classifying a dataset from Santander Bank and two datasets from UCI machine learning repository. The results show that our methods can achieve higher Area Under the Curve (AUC) and F-measure. We also compare them with filter-based feature selection approaches, i.e., Chi-Square and F-statistic. The results show that they outperform them but need slightly more computational efforts."]},{"key":"dc:title","label":"Title","values":["Decision tree rule-based feature selection for imbalanced data"]}]}],"canonical_facts":{"dc:contributor":["MengChu Zhou","Osvaldo Simeone","Yun Q. Shi"],"dc:creator":["Liu, Haoyue"],"dc:description.abstract":["A class imbalance problem appears in many real world applications, e.g., fault diagnosis, text categorization and fraud detection. When dealing with an imbalanced dataset, feature selection becomes an important issue. To address it, this work proposes a feature selection method that is based on a decision tree rule and weighted Gini index. The effectiveness of the proposed methods is verified by classifying a dataset from Santander Bank and two datasets from UCI machine learning repository. The results show that our methods can achieve higher Area Under the Curve (AUC) and F-measure. We also compare them with filter-based feature selection approaches, i.e., Chi-Square and F-statistic. The results show that they outperform them but need slightly more computational efforts."],"dc:identifier":["https://digitalcommons.njit.edu/theses/25"],"dc:subject":["Imbalanced data","Decision tree","Computer Engineering"],"dc:title":["Decision tree rule-based feature selection for imbalanced data"],"dc:type":["Thesis"],"thesis:degree_discipline":["Electrical and Computer Engineering"],"thesis:degree_name":["Master of Science in Computer Engineering - (M.S.)"]},"updated_at":"2026-07-24T03:22:01Z"}