{"id":{"repo_id":"umn","oai_identifier":"oai:conservancy.umn.edu:11299/199089"},"canonical_url":"https://search.dev.ndltd.org/etd/umn/oai:conservancy.umn.edu:11299/199089","repository":{"repo_id":"umn","name":"University of Minnesota","base_url":"https://conservancy.umn.edu/server/oai/request"},"display":{"title":"Statistical Methods for Large Complex Datasets","abstract":"Modern technological advancements have enabled massive-scale collection, processing and storage of information triggering the onset of the `big data' era where in every two days now we create as much data as we did in the entire twentieth century. This thesis aims at developing novel statistical methods that can efficiently analyze a variety of large complex datasets. Underlying the umbrella theme of big data modeling, we present statistical methods for two different classes of large complex datasets. The first half of the thesis focuses on the 'large n' problem for large spatial or spatio-temporal datasets where observations exhibit strong dependencies across space and time. In the second half of this thesis we present methods for high-dimensional regression in the `large p small n' setting for datasets that contain measurement errors or change points.","abstract_html":"Modern technological advancements have enabled massive-scale collection, processing and storage of information triggering the onset of the `big data&#x27; era where in every two days now we create as much data as we did in the entire twentieth century. This thesis aims at developing novel statistical methods that can efficiently analyze a variety of large complex datasets. Underlying the umbrella theme of big data modeling, we present statistical methods for two different classes of large complex datasets. The first half of the thesis focuses on the &#x27;large n&#x27; problem for large spatial or spatio-temporal datasets where observations exhibit strong dependencies across space and time. In the second half of this thesis we present methods for high-dimensional regression in the `large p small n&#x27; setting for datasets that contain measurement errors or change points.","abstract_has_math":false,"creators":["Datta, Abhirup"],"institution":null,"degree_name":null,"degree_level":null,"degree_discipline":null,"degree_department":null,"school":null,"contributors":[],"advisors":[],"committee_chairs":[],"committee_members":[],"year":2016,"date_issued":"2016-05","date_published":"2016-05","updated_at":"2026-07-24T05:20:01Z","subjects":["Big data","High dimensional data","Large spatial data"],"languages":["en"],"rights":[],"rights_urls":[],"identifier_entries":[]},"links":{"outbound_url":"http://hdl.handle.net/11299/199089","outbound_label":"Handle","outbound_source":"dc:identifier.uri"},"metadata_groups":[{"id":"people","label":"People","entries":[{"key":"dc:creator","label":"Author","values":["Datta, Abhirup"]}]},{"id":"academic_context","label":"Academic Context","entries":[{"key":"dc:date.accessioned","label":"Dc Date Accessioned","values":["2018-08-14T19:31:05Z"]},{"key":"dc:date.available","label":"Dc Date Available","values":["2018-08-14T19:31:05Z"]},{"key":"dc:date.issued","label":"Date","values":["2016-05"]},{"key":"dc:type","label":"Dc Type","values":["Thesis or Dissertation"]}]},{"id":"subjects_keywords","label":"Subjects and Keywords","entries":[{"key":"dc:subject","label":"Dc Subject","values":["Big data","High dimensional data","Large spatial data"]}]},{"id":"language_rights","label":"Language and Rights","entries":[{"key":"dc:language.iso","label":"Language (ISO)","values":["en"]}]},{"id":"identifiers","label":"Identifiers","entries":[{"key":"dc:identifier.uri","label":"Identifier URI","values":["http://hdl.handle.net/11299/199089"]}]},{"id":"additional","label":"Additional Metadata","entries":[{"key":"dc:description","label":"Description","values":["University of Minnesota Ph.D. dissertation. 2016. Major: Biostatistics. Advisors: Sudipto Banerjee, Hui Zou. 1 computer file (PDF); 175 pages."]},{"key":"dc:description.abstract","label":"Abstract","values":["Modern technological advancements have enabled massive-scale collection, processing and storage of information triggering the onset of the `big data' era where in every two days now we create as much data as we did in the entire twentieth century. This thesis aims at developing novel statistical methods that can efficiently analyze a variety of large complex datasets. Underlying the umbrella theme of big data modeling, we present statistical methods for two different classes of large complex datasets. The first half of the thesis focuses on the 'large n' problem for large spatial or spatio-temporal datasets where observations exhibit strong dependencies across space and time. In the second half of this thesis we present methods for high-dimensional regression in the `large p small n' setting for datasets that contain measurement errors or change points."]},{"key":"dc:title","label":"Title","values":["Statistical Methods for Large Complex Datasets"]}]}],"canonical_facts":{"dc:creator":["Datta, Abhirup"],"dc:date.accessioned":["2018-08-14T19:31:05Z"],"dc:date.available":["2018-08-14T19:31:05Z"],"dc:date.issued":["2016-05"],"dc:description":["University of Minnesota Ph.D. dissertation. 2016. Major: Biostatistics. Advisors: Sudipto Banerjee, Hui Zou. 1 computer file (PDF); 175 pages."],"dc:description.abstract":["Modern technological advancements have enabled massive-scale collection, processing and storage of information triggering the onset of the `big data' era where in every two days now we create as much data as we did in the entire twentieth century. This thesis aims at developing novel statistical methods that can efficiently analyze a variety of large complex datasets. Underlying the umbrella theme of big data modeling, we present statistical methods for two different classes of large complex datasets. The first half of the thesis focuses on the 'large n' problem for large spatial or spatio-temporal datasets where observations exhibit strong dependencies across space and time. In the second half of this thesis we present methods for high-dimensional regression in the `large p small n' setting for datasets that contain measurement errors or change points."],"dc:identifier.uri":["http://hdl.handle.net/11299/199089"],"dc:language.iso":["en"],"dc:subject":["Big data","High dimensional data","Large spatial data"],"dc:title":["Statistical Methods for Large Complex Datasets"],"dc:type":["Thesis or Dissertation"]},"updated_at":"2026-07-24T05:20:01Z"}