{"id":{"repo_id":"uiuc","oai_identifier":"oai:www.ideals.illinois.edu:2142/129340"},"canonical_url":"https://search.dev.ndltd.org/etd/uiuc/oai:www.ideals.illinois.edu:2142/129340","repository":{"repo_id":"uiuc","name":"University of Illinois - Urbana-Champaign","base_url":"https://www.ideals.illinois.edu/oai-pmh"},"display":{"title":"Optimization opportunities for various heterogeneous pipelines","abstract":"Submission original under an indefinite embargo labeled 'Open Access'. The submission was exported from vireo on 2025-10-19 without embargo terms","abstract_html":"Submission original under an indefinite embargo labeled &#x27;Open Access&#x27;. The submission was exported from vireo on 2025-10-19 without embargo terms","abstract_has_math":false,"creators":["Patel, Krut Sachindev"],"institution":"University of Illinois Urbana-Champaign","degree_name":"M.S.","degree_level":"Thesis","degree_discipline":"Computer Science","degree_department":null,"school":null,"contributors":["Mendis, Charith"],"advisors":[],"committee_chairs":[],"committee_members":[],"year":2025,"date_issued":"2025-05-08","date_published":"2025-05-08","updated_at":"2026-07-22T22:25:05Z","subjects":["Data Processing","Machine Learning","Heterogeneous Systems"],"languages":["en","eng"],"rights":["Copyright 2025 Krut Patel"],"rights_urls":[],"identifier_entries":[]},"links":{"outbound_url":"https://hdl.handle.net/2142/129340","outbound_label":"Handle","outbound_source":"dc:identifier"},"metadata_groups":[{"id":"people","label":"People","entries":[{"key":"dc:contributor","label":"Contributor","values":["Mendis, Charith"]},{"key":"dc:creator","label":"Author","values":["Patel, Krut Sachindev"]}]},{"id":"academic_context","label":"Academic Context","entries":[{"key":"dc:date","label":"Dc Date","values":["2025-05-08","2025-05"]},{"key":"dc:type","label":"Dc Type","values":["text"]},{"key":"thesis:degree_discipline","label":"Discipline","values":["Computer Science"]},{"key":"thesis:degree_level","label":"Degree Level","values":["Thesis"]},{"key":"thesis:degree_name","label":"Degree Name","values":["M.S."]},{"key":"thesis:institution_name","label":"Thesis Institution Name","values":["University of Illinois Urbana-Champaign"]}]},{"id":"subjects_keywords","label":"Subjects and Keywords","entries":[{"key":"dc:subject","label":"Dc Subject","values":["Data Processing","Machine Learning","Heterogeneous Systems"]}]},{"id":"language_rights","label":"Language and Rights","entries":[{"key":"dc:language","label":"Dc Language","values":["en","eng"]},{"key":"dc:rights","label":"Dc Rights","values":["Copyright 2025 Krut Patel"]}]},{"id":"identifiers","label":"Identifiers","entries":[{"key":"dc:identifier","label":"Identifier","values":["https://hdl.handle.net/2142/129340"]}]},{"id":"additional","label":"Additional Metadata","entries":[{"key":"dc:description","label":"Description","values":["Submission original under an indefinite embargo labeled 'Open Access'. The submission was exported from vireo on 2025-10-19 without embargo terms","The student, Krut Patel, accepted the attached license on 2025-05-07 at 17:07.","The student, Krut Patel, submitted this Thesis for approval on 2025-05-07 at 18:04.","This Thesis was approved for publication on 2025-05-08 at 16:25.","DSpace SAF Submission Ingestion Package generated from Vireo submission #22255 on 2025-10-19 at 18:13:48","Data preprocessing is an important step in machine learning workloads, and has started to take up increasingly higher share of the total run times. Frameworks such as tf.data and DataJuicer have gained popularity because they provide simple abstractions to define and potentially parallelize the preprocessing pipelines. More recent works have explored further optimization opportunities, focusing on offloading the computation to separate devices, and reordering the operators to reduce data transfer costs. However, they require manual input from the users for determining the data dependencies between various operators. This thesis focuses on studying and uncovering optimization opportunities for preprocessing pipelines. Guided by detailed performance profiling, we investigate the exact conditions under which reordering can provide benefits. We also analyze the impact of jointly optimizing reordering and device placement of the operators for better performance. Additionally, we explore the possibility of using automated search of the possible optimizations by using the ML model itself as a search metric. Moreover, we investigate a data pipeline from a state of the art multimodal model and detail the novel features of its performance characteristics. Finally, we conclude with a description of the open problems that need to be tackled for building data preprocessing systems for modern machine learning workloads."]},{"key":"dc:format","label":"Dc Format","values":["application/pdf"]},{"key":"dc:title","label":"Title","values":["Optimization opportunities for various heterogeneous pipelines"]}]}],"canonical_facts":{"dc:contributor":["Mendis, Charith"],"dc:creator":["Patel, Krut Sachindev"],"dc:date":["2025-05-08","2025-05"],"dc:description":["Submission original under an indefinite embargo labeled 'Open Access'. The submission was exported from vireo on 2025-10-19 without embargo terms","The student, Krut Patel, accepted the attached license on 2025-05-07 at 17:07.","The student, Krut Patel, submitted this Thesis for approval on 2025-05-07 at 18:04.","This Thesis was approved for publication on 2025-05-08 at 16:25.","DSpace SAF Submission Ingestion Package generated from Vireo submission #22255 on 2025-10-19 at 18:13:48","Data preprocessing is an important step in machine learning workloads, and has started to take up increasingly higher share of the total run times. Frameworks such as tf.data and DataJuicer have gained popularity because they provide simple abstractions to define and potentially parallelize the preprocessing pipelines. More recent works have explored further optimization opportunities, focusing on offloading the computation to separate devices, and reordering the operators to reduce data transfer costs. However, they require manual input from the users for determining the data dependencies between various operators. This thesis focuses on studying and uncovering optimization opportunities for preprocessing pipelines. Guided by detailed performance profiling, we investigate the exact conditions under which reordering can provide benefits. We also analyze the impact of jointly optimizing reordering and device placement of the operators for better performance. Additionally, we explore the possibility of using automated search of the possible optimizations by using the ML model itself as a search metric. Moreover, we investigate a data pipeline from a state of the art multimodal model and detail the novel features of its performance characteristics. Finally, we conclude with a description of the open problems that need to be tackled for building data preprocessing systems for modern machine learning workloads."],"dc:format":["application/pdf"],"dc:identifier":["https://hdl.handle.net/2142/129340"],"dc:language":["en","eng"],"dc:rights":["Copyright 2025 Krut Patel"],"dc:subject":["Data Processing","Machine Learning","Heterogeneous Systems"],"dc:title":["Optimization opportunities for various heterogeneous pipelines"],"dc:type":["text"],"thesis:degree_discipline":["Computer Science"],"thesis:degree_level":["Thesis"],"thesis:degree_name":["M.S."],"thesis:institution_name":["University of Illinois Urbana-Champaign"]},"updated_at":"2026-07-22T22:25:05Z"}