{"id":{"repo_id":"vcu","oai_identifier":"oai:scholarscompass.vcu.edu:etd-2432"},"canonical_url":"https://search.dev.ndltd.org/etd/vcu/oai:scholarscompass.vcu.edu:etd-2432","repository":{"repo_id":"vcu","name":"Virginia Commonwealth University","base_url":"https://scholarscompass.vcu.edu/do/oai/"},"display":{"title":"Quantifying the Effects of Correlated Covariates on Variable Importance Estimates from Random Forests","abstract":"Recent advances in computing technology have lead to the development of algorithmic modeling techniques. These methods can be used to analyze data which are difficult to analyze using traditional statistical models. This study examined the effectiveness of variable importance estimates from the random forest algorithm in identifying the true predictor among a large number of candidate predictors. A simulation study was conducted using twenty different levels of association among the independent variables and seven different levels of association between the true predictor and the response. We conclude that the random forest method is an effective classification tool when the goals of a study are to produce an accurate classifier and to provide insight regarding the discriminative ability of individual predictor variables. These goals are common in gene expression analysis, therefore we apply the random forest method for the purpose of estimating variable importance on a microarray data set.","abstract_html":"Recent advances in computing technology have lead to the development of algorithmic modeling techniques. These methods can be used to analyze data which are difficult to analyze using traditional statistical models. This study examined the effectiveness of variable importance estimates from the random forest algorithm in identifying the true predictor among a large number of candidate predictors. A simulation study was conducted using twenty different levels of association among the independent variables and seven different levels of association between the true predictor and the response. We conclude that the random forest method is an effective classification tool when the goals of a study are to produce an accurate classifier and to provide insight regarding the discriminative ability of individual predictor variables. These goals are common in gene expression analysis, therefore we apply the random forest method for the purpose of estimating variable importance on a microarray data set.","abstract_has_math":false,"creators":["Kimes, Ryan Vincent"],"institution":null,"degree_name":"Master of Science","degree_level":"Thesis","degree_discipline":"Biostatistics","degree_department":null,"school":null,"contributors":["Dr. Kellie J. Archer"],"advisors":[],"committee_chairs":[],"committee_members":[],"year":2006,"date_issued":"2006-01-01T08:00:00Z","date_published":"2006-01-01T08:00:00Z","updated_at":"2026-07-24T05:55:25Z","subjects":["algorithmic modeling","data analysis","algorithm","random forest method","Biostatistics","Physical Sciences and Mathematics","Statistics and Probability"],"languages":[],"rights":["© The Author"],"rights_urls":[],"identifier_entries":[{"key":"dc:identifier","label":"Identifier","values":["https://scholarscompass.vcu.edu/etd/1433"],"render_values":[{"text":"https://scholarscompass.vcu.edu/etd/1433","href":"https://scholarscompass.vcu.edu/etd/1433","code":true}]}]},"links":{"outbound_url":"https://doi.org/10.25772/GYCH-0G22","outbound_label":"DOI","outbound_source":"dc:identifier"},"metadata_groups":[{"id":"people","label":"People","entries":[{"key":"dc:contributor","label":"Contributor","values":["Dr. Kellie J. Archer"]},{"key":"dc:creator","label":"Author","values":["Kimes, Ryan Vincent"]}]},{"id":"academic_context","label":"Academic Context","entries":[{"key":"dc:date.available","label":"Dc Date Available","values":["2014-07-09T07:00:00Z"]},{"key":"thesis:degree_discipline","label":"Discipline","values":["Biostatistics"]},{"key":"thesis:degree_level","label":"Degree Level","values":["Thesis"]},{"key":"thesis:degree_name","label":"Degree Name","values":["Master of Science"]}]},{"id":"subjects_keywords","label":"Subjects and Keywords","entries":[{"key":"dc:subject","label":"Dc Subject","values":["algorithmic modeling","data analysis","algorithm","random forest method","Biostatistics","Physical Sciences and Mathematics","Statistics and Probability"]}]},{"id":"language_rights","label":"Language and Rights","entries":[{"key":"dc:rights","label":"Dc Rights","values":["© The Author"]}]},{"id":"identifiers","label":"Identifiers","entries":[{"key":"dc:identifier","label":"Identifier","values":["https://doi.org/10.25772/GYCH-0G22","https://scholarscompass.vcu.edu/etd/1433"]}]},{"id":"additional","label":"Additional Metadata","entries":[{"key":"dc:description.abstract","label":"Abstract","values":["Recent advances in computing technology have lead to the development of algorithmic modeling techniques. These methods can be used to analyze data which are difficult to analyze using traditional statistical models. This study examined the effectiveness of variable importance estimates from the random forest algorithm in identifying the true predictor among a large number of candidate predictors. A simulation study was conducted using twenty different levels of association among the independent variables and seven different levels of association between the true predictor and the response. We conclude that the random forest method is an effective classification tool when the goals of a study are to produce an accurate classifier and to provide insight regarding the discriminative ability of individual predictor variables. These goals are common in gene expression analysis, therefore we apply the random forest method for the purpose of estimating variable importance on a microarray data set."]},{"key":"dc:title","label":"Title","values":["Quantifying the Effects of Correlated Covariates on Variable Importance Estimates from Random Forests"]}]}],"canonical_facts":{"dc:contributor":["Dr. Kellie J. Archer"],"dc:creator":["Kimes, Ryan Vincent"],"dc:date.available":["2014-07-09T07:00:00Z"],"dc:description.abstract":["Recent advances in computing technology have lead to the development of algorithmic modeling techniques. These methods can be used to analyze data which are difficult to analyze using traditional statistical models. This study examined the effectiveness of variable importance estimates from the random forest algorithm in identifying the true predictor among a large number of candidate predictors. A simulation study was conducted using twenty different levels of association among the independent variables and seven different levels of association between the true predictor and the response. We conclude that the random forest method is an effective classification tool when the goals of a study are to produce an accurate classifier and to provide insight regarding the discriminative ability of individual predictor variables. These goals are common in gene expression analysis, therefore we apply the random forest method for the purpose of estimating variable importance on a microarray data set."],"dc:identifier":["https://doi.org/10.25772/GYCH-0G22","https://scholarscompass.vcu.edu/etd/1433"],"dc:rights":["© The Author"],"dc:subject":["algorithmic modeling","data analysis","algorithm","random forest method","Biostatistics","Physical Sciences and Mathematics","Statistics and Probability"],"dc:title":["Quantifying the Effects of Correlated Covariates on Variable Importance Estimates from Random Forests"],"thesis:degree_discipline":["Biostatistics"],"thesis:degree_level":["Thesis"],"thesis:degree_name":["Master of Science"]},"updated_at":"2026-07-24T05:55:25Z"}