{"id":{"repo_id":"penn","oai_identifier":"oai:repository.upenn.edu:20.500.14332/61229"},"canonical_url":"https://search.dev.ndltd.org/etd/penn/oai:repository.upenn.edu:20.500.14332/61229","repository":{"repo_id":"penn","name":"University of Pennsylvania","base_url":"https://repository.upenn.edu/server/oai/request"},"display":{"title":"MOVING BLACK-BOXING TOWARDS STATISTICS: CASE STUDIES FROM AMERICAN FOOTBALL","abstract":"Over the past decade, the explosion of publicly available data and off-the-shelf machine learning (ML) tools has popularized a common data science workflow: (1) obtain a dataset, (2) fit a black-box ML model, and (3) use its predictions. This workflow has become even more streamlined with LLMs—just upload your dataset to ChatGPT, and it will fit a model without requiring any specification. This paradigm is especially prevalent in sports analytics. While the modern ML pipeline excels in data-rich environments, it struggles with challenges that statisticians traditionally consider, such as limited data, selection bias, strong dependency structures, and the need for uncertainty quantification. These challenges are pervasive in sports analytics. Hence, we propose a shift in emphasis across data science away from the typical black-box machine learning workflow and towards an emphasis on statistical thinking. We illustrate our proposed emphasis through case studies from American football: expected points, win probability, and NFL draft position value curves.","abstract_html":"Over the past decade, the explosion of publicly available data and off-the-shelf machine learning (ML) tools has popularized a common data science workflow: (1) obtain a dataset, (2) fit a black-box ML model, and (3) use its predictions. This workflow has become even more streamlined with LLMs—just upload your dataset to ChatGPT, and it will fit a model without requiring any specification. This paradigm is especially prevalent in sports analytics. While the modern ML pipeline excels in data-rich environments, it struggles with challenges that statisticians traditionally consider, such as limited data, selection bias, strong dependency structures, and the need for uncertainty quantification. These challenges are pervasive in sports analytics. Hence, we propose a shift in emphasis across data science away from the typical black-box machine learning workflow and towards an emphasis on statistical thinking. We illustrate our proposed emphasis through case studies from American football: expected points, win probability, and NFL draft position value curves.","abstract_has_math":false,"creators":["Brill, Ryan"],"institution":null,"degree_name":null,"degree_level":null,"degree_discipline":null,"degree_department":null,"school":null,"contributors":[],"advisors":["Wyner, Abraham, J"],"committee_chairs":[],"committee_members":[],"year":2025,"date_issued":"2025","date_published":"2025","updated_at":"2026-07-24T03:47:40Z","subjects":["Statistics and Probability"],"languages":["en"],"rights":[],"rights_urls":[],"identifier_entries":[]},"links":{"outbound_url":"https://repository.upenn.edu/handle/20.500.14332/61229","outbound_label":"Repository record","outbound_source":"dc:identifier.uri"},"metadata_groups":[{"id":"people","label":"People","entries":[{"key":"dc:contributor.advisor","label":"Advisor","values":["Wyner, Abraham, J"]},{"key":"dc:creator","label":"Author","values":["Brill, Ryan"]}]},{"id":"academic_context","label":"Academic Context","entries":[{"key":"dc:date.accessioned","label":"Dc Date Accessioned","values":["2025-06-11T19:09:08Z"]},{"key":"dc:date.available","label":"Dc Date Available","values":["2025-06-11T19:09:08Z"]},{"key":"dc:date.issued","label":"Date","values":["2025"]},{"key":"dc:type","label":"Dc Type","values":["Dissertation/Thesis"]}]},{"id":"subjects_keywords","label":"Subjects and Keywords","entries":[{"key":"dc:subject","label":"Dc Subject","values":["Statistics and Probability"]}]},{"id":"language_rights","label":"Language and Rights","entries":[{"key":"dc:language.iso","label":"Language (ISO)","values":["en"]}]},{"id":"identifiers","label":"Identifiers","entries":[{"key":"dc:identifier.uri","label":"Identifier URI","values":["https://repository.upenn.edu/handle/20.500.14332/61229"]}]},{"id":"additional","label":"Additional Metadata","entries":[{"key":"dc:description.abstract","label":"Abstract","values":["Over the past decade, the explosion of publicly available data and off-the-shelf machine learning (ML) tools has popularized a common data science workflow: (1) obtain a dataset, (2) fit a black-box ML model, and (3) use its predictions. This workflow has become even more streamlined with LLMs—just upload your dataset to ChatGPT, and it will fit a model without requiring any specification. This paradigm is especially prevalent in sports analytics. While the modern ML pipeline excels in data-rich environments, it struggles with challenges that statisticians traditionally consider, such as limited data, selection bias, strong dependency structures, and the need for uncertainty quantification. These challenges are pervasive in sports analytics. Hence, we propose a shift in emphasis across data science away from the typical black-box machine learning workflow and towards an emphasis on statistical thinking. We illustrate our proposed emphasis through case studies from American football: expected points, win probability, and NFL draft position value curves."]},{"key":"dc:description.degree","label":"Dc Description Degree","values":["Doctor of Philosophy (PhD)"]},{"key":"dc:title","label":"Title","values":["MOVING BLACK-BOXING TOWARDS STATISTICS: CASE STUDIES FROM AMERICAN FOOTBALL"]}]}],"canonical_facts":{"dc:contributor.advisor":["Wyner, Abraham, J"],"dc:creator":["Brill, Ryan"],"dc:date.accessioned":["2025-06-11T19:09:08Z"],"dc:date.available":["2025-06-11T19:09:08Z"],"dc:date.issued":["2025"],"dc:description.abstract":["Over the past decade, the explosion of publicly available data and off-the-shelf machine learning (ML) tools has popularized a common data science workflow: (1) obtain a dataset, (2) fit a black-box ML model, and (3) use its predictions. This workflow has become even more streamlined with LLMs—just upload your dataset to ChatGPT, and it will fit a model without requiring any specification. This paradigm is especially prevalent in sports analytics. While the modern ML pipeline excels in data-rich environments, it struggles with challenges that statisticians traditionally consider, such as limited data, selection bias, strong dependency structures, and the need for uncertainty quantification. These challenges are pervasive in sports analytics. Hence, we propose a shift in emphasis across data science away from the typical black-box machine learning workflow and towards an emphasis on statistical thinking. We illustrate our proposed emphasis through case studies from American football: expected points, win probability, and NFL draft position value curves."],"dc:description.degree":["Doctor of Philosophy (PhD)"],"dc:identifier.uri":["https://repository.upenn.edu/handle/20.500.14332/61229"],"dc:language.iso":["en"],"dc:subject":["Statistics and Probability"],"dc:title":["MOVING BLACK-BOXING TOWARDS STATISTICS: CASE STUDIES FROM AMERICAN FOOTBALL"],"dc:type":["Dissertation/Thesis"]},"updated_at":"2026-07-24T03:47:40Z"}