{"id":{"repo_id":"mit","oai_identifier":"oai:dspace.mit.edu:1721.1/151389"},"canonical_url":"https://search.dev.ndltd.org/etd/mit/oai:dspace.mit.edu:1721.1/151389","repository":{"repo_id":"mit","name":"MIT","base_url":"https://dspace.mit.edu/oai/request"},"display":{"title":"Towards Creating Synthetic Data Testbeds for Research","abstract":"Insurance datasets are generally private in order to protect user information, making it difficult for the ML research community to access and experiment with this data. To increase accessibility and innovation on private insurance data, we compile and share publicly available insurance datasets, analyze challenges inherent in these datasets, and propose, motivate, and evaluate a Synthetic Data sharing framework called Synthetic Insurance Data (SID) Testbed that can be used to improve ML performance on tabular datasets by allowing collaborators to generate Synthetic Data for Data Augmentation. In addition to this framework, we recognize that tabular data augmentation is not a well understood phenomenon, and we run controlled experiments to better understand how and when data augmentation improves machine learning performance in the setting of tabular data.","abstract_html":"Insurance datasets are generally private in order to protect user information, making it difficult for the ML research community to access and experiment with this data. To increase accessibility and innovation on private insurance data, we compile and share publicly available insurance datasets, analyze challenges inherent in these datasets, and propose, motivate, and evaluate a Synthetic Data sharing framework called Synthetic Insurance Data (SID) Testbed that can be used to improve ML performance on tabular datasets by allowing collaborators to generate Synthetic Data for Data Augmentation. In addition to this framework, we recognize that tabular data augmentation is not a well understood phenomenon, and we run controlled experiments to better understand how and when data augmentation improves machine learning performance in the setting of tabular data.","abstract_has_math":false,"creators":["Oufattole, Nassim"],"institution":"Massachusetts Institute of Technology","degree_name":"Master","degree_level":null,"degree_discipline":null,"degree_department":"Massachusetts Institute of Technology. Department of Electrical Engineering and Computer Science","school":null,"contributors":[],"advisors":["Veeramachaneni, Kalyan"],"committee_chairs":[],"committee_members":[],"year":2023,"date_issued":"2023-06","date_published":"2023-06","updated_at":"2026-07-22T22:21:05Z","subjects":[],"languages":[],"rights":["In Copyright - Educational Use Permitted","Copyright retained by author(s)"],"rights_urls":["https://rightsstatements.org/page/InC-EDU/1.0/"],"identifier_entries":[]},"links":{"outbound_url":"https://hdl.handle.net/1721.1/151389","outbound_label":"Handle","outbound_source":"dc:identifier.uri"},"metadata_groups":[{"id":"people","label":"People","entries":[{"key":"dc:contributor.advisor","label":"Advisor","values":["Veeramachaneni, Kalyan"]},{"key":"dc:contributor.department","label":"Department","values":["Massachusetts Institute of Technology. Department of Electrical Engineering and Computer Science"]},{"key":"dc:creator","label":"Author","values":["Oufattole, Nassim"]}]},{"id":"academic_context","label":"Academic Context","entries":[{"key":"dc:date.accessioned","label":"Dc Date Accessioned","values":["2023-07-31T19:36:02Z"]},{"key":"dc:date.available","label":"Dc Date Available","values":["2023-07-31T19:36:02Z"]},{"key":"dc:date.issued","label":"Date","values":["2023-06"]},{"key":"dc:publisher","label":"Institution","values":["Massachusetts Institute of Technology"]},{"key":"dc:type","label":"Dc Type","values":["Thesis"]},{"key":"thesis:degree_name","label":"Degree Name","values":["Master","Master of Science in Electrical Engineering and Computer Science"]}]},{"id":"language_rights","label":"Language and Rights","entries":[{"key":"dc:rights","label":"Dc Rights","values":["In Copyright - Educational Use Permitted","Copyright retained by author(s)"]},{"key":"dc:rights.uri","label":"Rights URI","values":["https://rightsstatements.org/page/InC-EDU/1.0/"]}]},{"id":"identifiers","label":"Identifiers","entries":[{"key":"dc:identifier.uri","label":"Identifier URI","values":["https://hdl.handle.net/1721.1/151389"]}]},{"id":"additional","label":"Additional Metadata","entries":[{"key":"dc:description.abstract","label":"Abstract","values":["Insurance datasets are generally private in order to protect user information, making it difficult for the ML research community to access and experiment with this data. To increase accessibility and innovation on private insurance data, we compile and share publicly available insurance datasets, analyze challenges inherent in these datasets, and propose, motivate, and evaluate a Synthetic Data sharing framework called Synthetic Insurance Data (SID) Testbed that can be used to improve ML performance on tabular datasets by allowing collaborators to generate Synthetic Data for Data Augmentation. In addition to this framework, we recognize that tabular data augmentation is not a well understood phenomenon, and we run controlled experiments to better understand how and when data augmentation improves machine learning performance in the setting of tabular data."]},{"key":"dc:description.degree","label":"Dc Description Degree","values":["S.M."]},{"key":"dc:title","label":"Title","values":["Towards Creating Synthetic Data Testbeds for Research"]}]}],"canonical_facts":{"dc:contributor.advisor":["Veeramachaneni, Kalyan"],"dc:contributor.department":["Massachusetts Institute of Technology. Department of Electrical Engineering and Computer Science"],"dc:creator":["Oufattole, Nassim"],"dc:date.accessioned":["2023-07-31T19:36:02Z"],"dc:date.available":["2023-07-31T19:36:02Z"],"dc:date.issued":["2023-06"],"dc:description.abstract":["Insurance datasets are generally private in order to protect user information, making it difficult for the ML research community to access and experiment with this data. To increase accessibility and innovation on private insurance data, we compile and share publicly available insurance datasets, analyze challenges inherent in these datasets, and propose, motivate, and evaluate a Synthetic Data sharing framework called Synthetic Insurance Data (SID) Testbed that can be used to improve ML performance on tabular datasets by allowing collaborators to generate Synthetic Data for Data Augmentation. In addition to this framework, we recognize that tabular data augmentation is not a well understood phenomenon, and we run controlled experiments to better understand how and when data augmentation improves machine learning performance in the setting of tabular data."],"dc:description.degree":["S.M."],"dc:identifier.uri":["https://hdl.handle.net/1721.1/151389"],"dc:publisher":["Massachusetts Institute of Technology"],"dc:rights":["In Copyright - Educational Use Permitted","Copyright retained by author(s)"],"dc:rights.uri":["https://rightsstatements.org/page/InC-EDU/1.0/"],"dc:title":["Towards Creating Synthetic Data Testbeds for Research"],"dc:type":["Thesis"],"thesis:degree_name":["Master","Master of Science in Electrical Engineering and Computer Science"]},"updated_at":"2026-07-22T22:21:05Z"}