{"id":{"repo_id":"uiuc","oai_identifier":"oai:www.ideals.illinois.edu:2142/121563"},"canonical_url":"https://search.dev.ndltd.org/etd/uiuc/oai:www.ideals.illinois.edu:2142/121563","repository":{"repo_id":"uiuc","name":"University of Illinois - Urbana-Champaign","base_url":"https://www.ideals.illinois.edu/oai-pmh"},"display":{"title":"A faster reinforcement learning approach to efficient job scheduling in Apache Spark","abstract":"Submission original under an indefinite embargo labeled 'Open Access'. The submission was exported from vireo on 2023-12-04 without embargo terms","abstract_html":"Submission original under an indefinite embargo labeled &#x27;Open Access&#x27;. The submission was exported from vireo on 2023-12-04 without embargo terms","abstract_has_math":false,"creators":["Gertsman, Arkadiy"],"institution":"University of Illinois at Urbana-Champaign","degree_name":"M.S.","degree_level":"Thesis","degree_discipline":"Industrial Engineering","degree_department":null,"school":null,"contributors":["Nagi, Rakesh"],"advisors":[],"committee_chairs":[],"committee_members":[],"year":2023,"date_issued":"2023-08","date_published":"2023-08","updated_at":"2026-07-22T22:25:00Z","subjects":["Reinforcement Learning","Graph Neural Networks","Job Scheduling","Apache Spark"],"languages":["en","eng"],"rights":["Copyright 2023 Arkadiy Gertsman"],"rights_urls":[],"identifier_entries":[]},"links":{"outbound_url":"https://hdl.handle.net/2142/121563","outbound_label":"Handle","outbound_source":"dc:identifier"},"metadata_groups":[{"id":"people","label":"People","entries":[{"key":"dc:contributor","label":"Contributor","values":["Nagi, Rakesh"]},{"key":"dc:creator","label":"Author","values":["Gertsman, Arkadiy"]}]},{"id":"academic_context","label":"Academic Context","entries":[{"key":"dc:date","label":"Dc Date","values":["2023-08","2023-07-21"]},{"key":"dc:type","label":"Dc Type","values":["text"]},{"key":"thesis:degree_discipline","label":"Discipline","values":["Industrial Engineering"]},{"key":"thesis:degree_level","label":"Degree Level","values":["Thesis"]},{"key":"thesis:degree_name","label":"Degree Name","values":["M.S."]},{"key":"thesis:institution_name","label":"Thesis Institution Name","values":["University of Illinois at Urbana-Champaign"]}]},{"id":"subjects_keywords","label":"Subjects and Keywords","entries":[{"key":"dc:subject","label":"Dc Subject","values":["Reinforcement Learning","Graph Neural Networks","Job Scheduling","Apache Spark"]}]},{"id":"language_rights","label":"Language and Rights","entries":[{"key":"dc:language","label":"Dc Language","values":["en","eng"]},{"key":"dc:rights","label":"Dc Rights","values":["Copyright 2023 Arkadiy Gertsman"]}]},{"id":"identifiers","label":"Identifiers","entries":[{"key":"dc:identifier","label":"Identifier","values":["https://hdl.handle.net/2142/121563"]}]},{"id":"additional","label":"Additional Metadata","entries":[{"key":"dc:description","label":"Description","values":["Submission original under an indefinite embargo labeled 'Open Access'. The submission was exported from vireo on 2023-12-04 without embargo terms","The student, Arkadiy Gertsman, accepted the attached license on 2023-07-20 at 16:47.","The student, Arkadiy Gertsman, submitted this Thesis for approval on 2023-07-20 at 16:51.","This Thesis was approved for publication on 2023-07-21 at 10:34.","DSpace SAF Submission Ingestion Package generated from Vireo submission #19760 on 2023-12-04 at 17:03:37","Job scheduling problems have been widely studied in theoretical computer science and operations research, and are commonly encountered in applied settings such as computer systems, manufacturing, and construction. There are many variants of job scheduling, but they share a common goal: designat- ing jobs to run on a set of parallel machines at different times, such that the machines are efficiently utilized. This thesis focuses on job scheduling in the context of Apache SparkTM, a popular data analytics engine that harnesses the power of distributed computing. Job scheduling is central to Spark, as each Spark application needs a scheduler to orchestrate its job submissions. The basic scheduling rules provided by Spark work well on lighter workloads, but sophisticated scheduling algorithms can greatly increase cluster efficiency when workloads are heavier. Previous work has introduced such algorithms, some hand-tuned and others learned. This thesis thoroughly documents Dec- ima, the state-of-the-art, reinforcement-learned Spark job scheduler, includ- ing a close look into their simulator and model architectures, and a new SMDP formulation of the problem. This thesis also proposes Decima++, an update to Decima which improves scheduling performance and reduces training time by over 11×."]},{"key":"dc:format","label":"Dc Format","values":["application/pdf"]},{"key":"dc:title","label":"Title","values":["A faster reinforcement learning approach to efficient job scheduling in Apache Spark"]}]}],"canonical_facts":{"dc:contributor":["Nagi, Rakesh"],"dc:creator":["Gertsman, Arkadiy"],"dc:date":["2023-08","2023-07-21"],"dc:description":["Submission original under an indefinite embargo labeled 'Open Access'. The submission was exported from vireo on 2023-12-04 without embargo terms","The student, Arkadiy Gertsman, accepted the attached license on 2023-07-20 at 16:47.","The student, Arkadiy Gertsman, submitted this Thesis for approval on 2023-07-20 at 16:51.","This Thesis was approved for publication on 2023-07-21 at 10:34.","DSpace SAF Submission Ingestion Package generated from Vireo submission #19760 on 2023-12-04 at 17:03:37","Job scheduling problems have been widely studied in theoretical computer science and operations research, and are commonly encountered in applied settings such as computer systems, manufacturing, and construction. There are many variants of job scheduling, but they share a common goal: designat- ing jobs to run on a set of parallel machines at different times, such that the machines are efficiently utilized. This thesis focuses on job scheduling in the context of Apache SparkTM, a popular data analytics engine that harnesses the power of distributed computing. Job scheduling is central to Spark, as each Spark application needs a scheduler to orchestrate its job submissions. The basic scheduling rules provided by Spark work well on lighter workloads, but sophisticated scheduling algorithms can greatly increase cluster efficiency when workloads are heavier. Previous work has introduced such algorithms, some hand-tuned and others learned. This thesis thoroughly documents Dec- ima, the state-of-the-art, reinforcement-learned Spark job scheduler, includ- ing a close look into their simulator and model architectures, and a new SMDP formulation of the problem. This thesis also proposes Decima++, an update to Decima which improves scheduling performance and reduces training time by over 11×."],"dc:format":["application/pdf"],"dc:identifier":["https://hdl.handle.net/2142/121563"],"dc:language":["en","eng"],"dc:rights":["Copyright 2023 Arkadiy Gertsman"],"dc:subject":["Reinforcement Learning","Graph Neural Networks","Job Scheduling","Apache Spark"],"dc:title":["A faster reinforcement learning approach to efficient job scheduling in Apache Spark"],"dc:type":["text"],"thesis:degree_discipline":["Industrial Engineering"],"thesis:degree_level":["Thesis"],"thesis:degree_name":["M.S."],"thesis:institution_name":["University of Illinois at Urbana-Champaign"]},"updated_at":"2026-07-22T22:25:00Z"}