{"id":{"repo_id":"nus","oai_identifier":"oai:scholarbank.nus.edu.sg:10635/317571"},"canonical_url":"https://search.dev.ndltd.org/etd/nus/oai:scholarbank.nus.edu.sg:10635/317571","repository":{"repo_id":"nus","name":"National University of Singapore","base_url":"https://scholarbank.nus.edu.sg/oai/request"},"display":{"title":"EFFECTIVE TRAINING OF NEURAL NETWORKS FOR BETTER GENERALIZATION","abstract":"Deep learning has achieved remarkable success, yet training deep neural networks remains costly, unstable, and poorly understood in terms of generalization. This thesis aims to make training more efficient and generalization-aware, addressing from optimization and data perspectives. From the optimization side, we develop practical algorithms that improve convergence speed and stability. We propose DRAG, a dimension-reduced adaptive gradient method that unifies the benefits of SGD and Adam, and a memory-efficient Shampoo using 4-bit Cholesky quantization with error feedback, enabling scalable second-order training. From the data side, we investigate why data-centric strategies enhance generalization. We analyze semi-supervised learning and data augmentation through the lens of feature learning, uncovering how they promote semantic diversity and robustness, and propose improved variants such as SA-FixMatch.","abstract_html":"Deep learning has achieved remarkable success, yet training deep neural networks remains costly, unstable, and poorly understood in terms of generalization. This thesis aims to make training more efficient and generalization-aware, addressing from optimization and data perspectives. From the optimization side, we develop practical algorithms that improve convergence speed and stability. We propose DRAG, a dimension-reduced adaptive gradient method that unifies the benefits of SGD and Adam, and a memory-efficient Shampoo using 4-bit Cholesky quantization with error feedback, enabling scalable second-order training. From the data side, we investigate why data-centric strategies enhance generalization. We analyze semi-supervised learning and data augmentation through the lens of feature learning, uncovering how they promote semantic diversity and robustness, and propose improved variants such as SA-FixMatch.","abstract_has_math":false,"creators":["LI JINGYANG"],"institution":null,"degree_name":null,"degree_level":null,"degree_discipline":null,"degree_department":null,"school":null,"contributors":[],"advisors":[],"committee_chairs":[],"committee_members":[],"year":2025,"date_issued":"2025-08-11","date_published":"2025-08-11","updated_at":"2026-07-24T03:33:34Z","subjects":["Generalization Theory","Data Augmentation","Semi-Supervised Learning","Second-Order Optimization","Adaptive Gradient Method","Neural Network Training"],"languages":[],"rights":[],"rights_urls":["https://scholarbank.nus.edu.sg/bitstreams/8a2b79bd-b50f-4be6-ab3b-e62385082d99/download"],"identifier_entries":[]},"links":{"outbound_url":null,"outbound_label":null,"outbound_source":null},"metadata_groups":[{"id":"people","label":"People","entries":[{"key":"dc:creator","label":"Author","values":["LI JINGYANG"]}]},{"id":"academic_context","label":"Academic Context","entries":[{"key":"dc:date.issued","label":"Date","values":["2025-08-11"]},{"key":"dc:relation.isreferencedby","label":"Dc Relation Isreferencedby","values":["https://scholarbank.nus.edu.sg/handle/10635/317571"]},{"key":"dc:type","label":"Dc Type","values":["Thesis"]}]},{"id":"subjects_keywords","label":"Subjects and Keywords","entries":[{"key":"dc:subject","label":"Dc Subject","values":["Generalization Theory","Data Augmentation","Semi-Supervised Learning","Second-Order Optimization","Adaptive Gradient Method","Neural Network Training"]}]},{"id":"language_rights","label":"Language and Rights","entries":[{"key":"dc:rights","label":"Dc Rights","values":["https://scholarbank.nus.edu.sg/bitstreams/8a2b79bd-b50f-4be6-ab3b-e62385082d99/download"]}]},{"id":"identifiers","label":"Identifiers","entries":[{"key":"dc:identifier.uri","label":"Identifier URI","values":["https://scholarbank.nus.edu.sg/bitstreams/a16b9c16-fe7b-49eb-ab88-b6f7d867d8d6/download"]}]},{"id":"additional","label":"Additional Metadata","entries":[{"key":"dc:description.abstract","label":"Abstract","values":["Deep learning has achieved remarkable success, yet training deep neural networks remains costly, unstable, and poorly understood in terms of generalization. This thesis aims to make training more efficient and generalization-aware, addressing from optimization and data perspectives. From the optimization side, we develop practical algorithms that improve convergence speed and stability. We propose DRAG, a dimension-reduced adaptive gradient method that unifies the benefits of SGD and Adam, and a memory-efficient Shampoo using 4-bit Cholesky quantization with error feedback, enabling scalable second-order training. From the data side, we investigate why data-centric strategies enhance generalization. We analyze semi-supervised learning and data augmentation through the lens of feature learning, uncovering how they promote semantic diversity and robustness, and propose improved variants such as SA-FixMatch."]},{"key":"dc:format.checksum.md5","label":"Dc Format Checksum Md5","values":["9f3c6f2aa8f96291ef77b0a18484521d","2d92458175888e8f83cb84e405271e49","5ad436da51e223a7878c5208b0687adc"]},{"key":"dc:title","label":"Title","values":["EFFECTIVE TRAINING OF NEURAL NETWORKS FOR BETTER GENERALIZATION"]}]}],"canonical_facts":{"dc:creator":["LI JINGYANG"],"dc:date.issued":["2025-08-11"],"dc:description.abstract":["Deep learning has achieved remarkable success, yet training deep neural networks remains costly, unstable, and poorly understood in terms of generalization. This thesis aims to make training more efficient and generalization-aware, addressing from optimization and data perspectives. From the optimization side, we develop practical algorithms that improve convergence speed and stability. We propose DRAG, a dimension-reduced adaptive gradient method that unifies the benefits of SGD and Adam, and a memory-efficient Shampoo using 4-bit Cholesky quantization with error feedback, enabling scalable second-order training. From the data side, we investigate why data-centric strategies enhance generalization. We analyze semi-supervised learning and data augmentation through the lens of feature learning, uncovering how they promote semantic diversity and robustness, and propose improved variants such as SA-FixMatch."],"dc:format.checksum.md5":["9f3c6f2aa8f96291ef77b0a18484521d","2d92458175888e8f83cb84e405271e49","5ad436da51e223a7878c5208b0687adc"],"dc:identifier.uri":["https://scholarbank.nus.edu.sg/bitstreams/a16b9c16-fe7b-49eb-ab88-b6f7d867d8d6/download"],"dc:relation.isreferencedby":["https://scholarbank.nus.edu.sg/handle/10635/317571"],"dc:rights":["https://scholarbank.nus.edu.sg/bitstreams/8a2b79bd-b50f-4be6-ab3b-e62385082d99/download"],"dc:subject":["Generalization Theory","Data Augmentation","Semi-Supervised Learning","Second-Order Optimization","Adaptive Gradient Method","Neural Network Training"],"dc:title":["EFFECTIVE TRAINING OF NEURAL NETWORKS FOR BETTER GENERALIZATION"],"dc:type":["Thesis"]},"updated_at":"2026-07-24T03:33:34Z"}