{"id":{"repo_id":"guelph","oai_identifier":"oai:atrium.lib.uoguelph.ca:10214/13025"},"canonical_url":"https://search.dev.ndltd.org/etd/guelph/oai:atrium.lib.uoguelph.ca:10214/13025","repository":{"repo_id":"guelph","name":"University of Guelph","base_url":"https://atrium.lib.uoguelph.ca/server/oai/request"},"display":{"title":"Bayesian Clustering Approaches for Discrete Data","abstract":"Unsupervised classification or clustering uses no a priori knowledge of the labels of the observations in the process of categorizing data. The research contained in this thesis focuses on the machine learning of discrete-valued gene expression datasets using clustering, with the aim of identifying gene co-expression networks. Specifically, a number of topics surrounding the use of mixture models and Markov chain Monte Carlo (MCMC) methods in clustering of discrete data from high-throughput transcriptome sequencing technologies is presented. After outlining current challenges and gaps in research with respect to clustering approaches, three mixture model-based clustering methods are presented: mixtures of multivariate Poisson-log normal distributions, mixtures of multivariate Poisson-log normal factor analyzers and mixtures of matrix-variate Poisson-log normal distributions. Significance, innovation, limitations and a number of future directions stemming from this research are discussed.","abstract_html":"Unsupervised classification or clustering uses no a priori knowledge of the labels of the observations in the process of categorizing data. The research contained in this thesis focuses on the machine learning of discrete-valued gene expression datasets using clustering, with the aim of identifying gene co-expression networks. Specifically, a number of topics surrounding the use of mixture models and Markov chain Monte Carlo (MCMC) methods in clustering of discrete data from high-throughput transcriptome sequencing technologies is presented. After outlining current challenges and gaps in research with respect to clustering approaches, three mixture model-based clustering methods are presented: mixtures of multivariate Poisson-log normal distributions, mixtures of multivariate Poisson-log normal factor analyzers and mixtures of matrix-variate Poisson-log normal distributions. Significance, innovation, limitations and a number of future directions stemming from this research are discussed.","abstract_has_math":false,"creators":["Silva, H. Anjali"],"institution":"University of Guelph","degree_name":null,"degree_level":null,"degree_discipline":null,"degree_department":null,"school":null,"contributors":[],"advisors":["Rothstein, Steven"],"committee_chairs":[],"committee_members":[],"year":2017,"date_issued":"2017-11-30","date_published":"2017-11-30","updated_at":"2026-08-21T16:45:04Z","subjects":["Clustering","RNA sequencing","Discrete data","Multivariate Poisson-Log Normal distribution","Markov chain Monte Carlo","Factor analyzers","Matrix variate distribution","Co-expression network"],"languages":["en"],"rights":[],"rights_urls":[],"identifier_entries":[]},"links":{"outbound_url":"http://hdl.handle.net/10214/13025","outbound_label":"Handle","outbound_source":"dc:identifier.uri"},"source_record":{"url":"https://atrium.lib.uoguelph.ca/server/oai/request?verb=GetRecord&metadataPrefix=dim&identifier=oai%3Aatrium.lib.uoguelph.ca%3A10214%2F13025","prefix":"dim"},"metadata_groups":[{"id":"people","label":"People","entries":[{"key":"dc:contributor.advisor","label":"Advisor","values":["Rothstein, Steven"]},{"key":"dc:creator","label":"Author","values":["Silva, H. Anjali"]}]},{"id":"academic_context","label":"Academic Context","entries":[{"key":"dc:date.accessioned","label":"Dc Date Accessioned","values":["2018-05-09T17:49:45Z"]},{"key":"dc:date.available","label":"Dc Date Available","values":["2020-06-19T05:00:51Z"]},{"key":"dc:date.issued","label":"Date","values":["2017-11-30"]},{"key":"dc:publisher","label":"Institution","values":["University of Guelph"]},{"key":"dc:type","label":"Dc Type","values":["Thesis"]}]},{"id":"subjects_keywords","label":"Subjects and Keywords","entries":[{"key":"dc:subject","label":"Dc Subject","values":["Clustering","RNA sequencing","Discrete data","Multivariate Poisson-Log Normal distribution","Markov chain Monte Carlo","Factor analyzers","Matrix variate distribution","Co-expression network"]}]},{"id":"language_rights","label":"Language and Rights","entries":[{"key":"dc:language.iso","label":"Language (ISO)","values":["en"]}]},{"id":"identifiers","label":"Identifiers","entries":[{"key":"dc:identifier.uri","label":"Identifier URI","values":["http://hdl.handle.net/10214/13025"]}]},{"id":"additional","label":"Additional Metadata","entries":[{"key":"dc:description.abstract","label":"Abstract","values":["Unsupervised classification or clustering uses no a priori knowledge of the labels of the observations in the process of categorizing data. The research contained in this thesis focuses on the machine learning of discrete-valued gene expression datasets using clustering, with the aim of identifying gene co-expression networks. Specifically, a number of topics surrounding the use of mixture models and Markov chain Monte Carlo (MCMC) methods in clustering of discrete data from high-throughput transcriptome sequencing technologies is presented. After outlining current challenges and gaps in research with respect to clustering approaches, three mixture model-based clustering methods are presented: mixtures of multivariate Poisson-log normal distributions, mixtures of multivariate Poisson-log normal factor analyzers and mixtures of matrix-variate Poisson-log normal distributions. Significance, innovation, limitations and a number of future directions stemming from this research are discussed."]},{"key":"dc:title","label":"Title","values":["Bayesian Clustering Approaches for Discrete Data"]}]}],"canonical_facts":{"dc:contributor.advisor":["Rothstein, Steven"],"dc:creator":["Silva, H. Anjali"],"dc:date.accessioned":["2018-05-09T17:49:45Z"],"dc:date.available":["2020-06-19T05:00:51Z"],"dc:date.issued":["2017-11-30"],"dc:description.abstract":["Unsupervised classification or clustering uses no a priori knowledge of the labels of the observations in the process of categorizing data. The research contained in this thesis focuses on the machine learning of discrete-valued gene expression datasets using clustering, with the aim of identifying gene co-expression networks. Specifically, a number of topics surrounding the use of mixture models and Markov chain Monte Carlo (MCMC) methods in clustering of discrete data from high-throughput transcriptome sequencing technologies is presented. After outlining current challenges and gaps in research with respect to clustering approaches, three mixture model-based clustering methods are presented: mixtures of multivariate Poisson-log normal distributions, mixtures of multivariate Poisson-log normal factor analyzers and mixtures of matrix-variate Poisson-log normal distributions. Significance, innovation, limitations and a number of future directions stemming from this research are discussed."],"dc:identifier.uri":["http://hdl.handle.net/10214/13025"],"dc:language.iso":["en"],"dc:publisher":["University of Guelph"],"dc:subject":["Clustering","RNA sequencing","Discrete data","Multivariate Poisson-Log Normal distribution","Markov chain Monte Carlo","Factor analyzers","Matrix variate distribution","Co-expression network"],"dc:title":["Bayesian Clustering Approaches for Discrete Data"],"dc:type":["Thesis"]},"updated_at":"2026-08-21T16:45:04Z"}