{"id":{"repo_id":"uiuc","oai_identifier":"oai:www.ideals.illinois.edu:2142/121497"},"canonical_url":"https://search.dev.ndltd.org/etd/uiuc/oai:www.ideals.illinois.edu:2142/121497","repository":{"repo_id":"uiuc","name":"University of Illinois - Urbana-Champaign","base_url":"https://www.ideals.illinois.edu/oai-pmh"},"display":{"title":"Multimodal spoken unit discovery with paired and unpaired modalities","abstract":"Submission original under an indefinite embargo labeled 'Open Access'. The submission was exported from vireo on 2023-12-04 without embargo terms","abstract_html":"Submission original under an indefinite embargo labeled &#x27;Open Access&#x27;. The submission was exported from vireo on 2023-12-04 without embargo terms","abstract_has_math":false,"creators":["Wang, Liming"],"institution":"University of Illinois at Urbana-Champaign","degree_name":"Ph.D.","degree_level":"Dissertation","degree_discipline":"Electrical & Computer Engr","degree_department":null,"school":null,"contributors":["Hasegawa-Johnson, Mark","Smaragdis, Paris","Schwing, Alexander","Fleck, Margaret"],"advisors":[],"committee_chairs":[],"committee_members":[],"year":2023,"date_issued":"2023-08","date_published":"2023-08","updated_at":"2026-07-22T22:24:57Z","subjects":["Acoustic Unit Discovery","Low-resource Speech Recognition","Unsupervised Speech Recognition","Multimodal Learning","Self-supervised Learning","Language Acquisition"],"languages":["en","eng"],"rights":["Copyright 2023 by Liming Wang"],"rights_urls":[],"identifier_entries":[]},"links":{"outbound_url":"https://hdl.handle.net/2142/121497","outbound_label":"Handle","outbound_source":"dc:identifier"},"metadata_groups":[{"id":"people","label":"People","entries":[{"key":"dc:contributor","label":"Contributor","values":["Hasegawa-Johnson, Mark","Smaragdis, Paris","Schwing, Alexander","Fleck, Margaret"]},{"key":"dc:creator","label":"Author","values":["Wang, Liming"]}]},{"id":"academic_context","label":"Academic Context","entries":[{"key":"dc:date","label":"Dc Date","values":["2023-08","2023-07-12"]},{"key":"dc:type","label":"Dc Type","values":["text"]},{"key":"thesis:degree_discipline","label":"Discipline","values":["Electrical & Computer Engr"]},{"key":"thesis:degree_level","label":"Degree Level","values":["Dissertation"]},{"key":"thesis:degree_name","label":"Degree Name","values":["Ph.D."]},{"key":"thesis:institution_name","label":"Thesis Institution Name","values":["University of Illinois at Urbana-Champaign"]}]},{"id":"subjects_keywords","label":"Subjects and Keywords","entries":[{"key":"dc:subject","label":"Dc Subject","values":["Acoustic Unit Discovery","Low-resource Speech Recognition","Unsupervised Speech Recognition","Multimodal Learning","Self-supervised Learning","Language Acquisition"]}]},{"id":"language_rights","label":"Language and Rights","entries":[{"key":"dc:language","label":"Dc Language","values":["en","eng"]},{"key":"dc:rights","label":"Dc Rights","values":["Copyright 2023 by Liming Wang"]}]},{"id":"identifiers","label":"Identifiers","entries":[{"key":"dc:identifier","label":"Identifier","values":["https://hdl.handle.net/2142/121497"]}]},{"id":"additional","label":"Additional Metadata","entries":[{"key":"dc:description","label":"Description","values":["Submission original under an indefinite embargo labeled 'Open Access'. The submission was exported from vireo on 2023-12-04 without embargo terms","The student, Liming Wang, accepted the attached license on 2023-07-10 at 18:16.","The student, Liming Wang, submitted this Dissertation for approval on 2023-07-10 at 19:58.","This Dissertation was approved for publication on 2023-07-12 at 10:20.","DSpace SAF Submission Ingestion Package generated from Vireo submission #19612 on 2023-12-04 at 17:01:43","This thesis addresses the challenge of low-resource speech recognition by formulating it as a multimodal learning problem. The goal is to build a multimodal spoken unit discovery system that does not require any textual transcripts. Instead, it leverages speech and semantically related, multimodal signals such as paired images, unpaired text and unpaired sign language videos. To this end, this thesis proposes several novel algorithms based on neural networks and probabilistic graphical models. Further, it provides theoretical insights and empirical evidence to validate the efficacy of multimodal signals for spoken unit discovery."]},{"key":"dc:format","label":"Dc Format","values":["application/pdf"]},{"key":"dc:title","label":"Title","values":["Multimodal spoken unit discovery with paired and unpaired modalities"]}]}],"canonical_facts":{"dc:contributor":["Hasegawa-Johnson, Mark","Smaragdis, Paris","Schwing, Alexander","Fleck, Margaret"],"dc:creator":["Wang, Liming"],"dc:date":["2023-08","2023-07-12"],"dc:description":["Submission original under an indefinite embargo labeled 'Open Access'. The submission was exported from vireo on 2023-12-04 without embargo terms","The student, Liming Wang, accepted the attached license on 2023-07-10 at 18:16.","The student, Liming Wang, submitted this Dissertation for approval on 2023-07-10 at 19:58.","This Dissertation was approved for publication on 2023-07-12 at 10:20.","DSpace SAF Submission Ingestion Package generated from Vireo submission #19612 on 2023-12-04 at 17:01:43","This thesis addresses the challenge of low-resource speech recognition by formulating it as a multimodal learning problem. The goal is to build a multimodal spoken unit discovery system that does not require any textual transcripts. Instead, it leverages speech and semantically related, multimodal signals such as paired images, unpaired text and unpaired sign language videos. To this end, this thesis proposes several novel algorithms based on neural networks and probabilistic graphical models. Further, it provides theoretical insights and empirical evidence to validate the efficacy of multimodal signals for spoken unit discovery."],"dc:format":["application/pdf"],"dc:identifier":["https://hdl.handle.net/2142/121497"],"dc:language":["en","eng"],"dc:rights":["Copyright 2023 by Liming Wang"],"dc:subject":["Acoustic Unit Discovery","Low-resource Speech Recognition","Unsupervised Speech Recognition","Multimodal Learning","Self-supervised Learning","Language Acquisition"],"dc:title":["Multimodal spoken unit discovery with paired and unpaired modalities"],"dc:type":["text"],"thesis:degree_discipline":["Electrical & Computer Engr"],"thesis:degree_level":["Dissertation"],"thesis:degree_name":["Ph.D."],"thesis:institution_name":["University of Illinois at Urbana-Champaign"]},"updated_at":"2026-07-22T22:24:57Z"}