{"id":{"repo_id":"uiuc","oai_identifier":"oai:www.ideals.illinois.edu:2142/117680"},"canonical_url":"https://search.dev.ndltd.org/etd/uiuc/oai:www.ideals.illinois.edu:2142/117680","repository":{"repo_id":"uiuc","name":"University of Illinois - Urbana-Champaign","base_url":"https://www.ideals.illinois.edu/oai-pmh"},"display":{"title":"SpeechSplit2: An efficient unsupervised speech disentanglement model for multi-aspect voice conversion","abstract":"Submission published under a 24 month embargo labeled 'U of I Access', the embargo will last until 2024-12-01","abstract_html":"Submission published under a 24 month embargo labeled &#x27;U of I Access&#x27;, the embargo will last until 2024-12-01","abstract_has_math":false,"creators":["Chan, Chak Ho"],"institution":"University of Illinois at Urbana-Champaign","degree_name":"M.S.","degree_level":"Thesis","degree_discipline":"Electrical & Computer Engr","degree_department":null,"school":null,"contributors":["Hasegawa-Johnson, Mark Allan"],"advisors":[],"committee_chairs":[],"committee_members":[],"year":2022,"date_issued":"2022-12","date_published":"2022-12","updated_at":"2026-07-22T22:24:56Z","subjects":["Unsupervised Learning","Voice Conversion","Speech Disentanglement","Signal Processing"],"languages":["en","eng"],"rights":["Copyright 2022 Chak Ho Chan"],"rights_urls":[],"identifier_entries":[]},"links":{"outbound_url":"https://hdl.handle.net/2142/117680","outbound_label":"Handle","outbound_source":"dc:identifier"},"metadata_groups":[{"id":"people","label":"People","entries":[{"key":"dc:contributor","label":"Contributor","values":["Hasegawa-Johnson, Mark Allan"]},{"key":"dc:creator","label":"Author","values":["Chan, Chak Ho"]}]},{"id":"academic_context","label":"Academic Context","entries":[{"key":"dc:date","label":"Dc Date","values":["2022-12","2022-12-07"]},{"key":"dc:type","label":"Dc Type","values":["text","Thesis"]},{"key":"thesis:degree_discipline","label":"Discipline","values":["Electrical & Computer Engr"]},{"key":"thesis:degree_level","label":"Degree Level","values":["Thesis"]},{"key":"thesis:degree_name","label":"Degree Name","values":["M.S."]},{"key":"thesis:institution_name","label":"Thesis Institution Name","values":["University of Illinois at Urbana-Champaign"]}]},{"id":"subjects_keywords","label":"Subjects and Keywords","entries":[{"key":"dc:subject","label":"Dc Subject","values":["Unsupervised Learning","Voice Conversion","Speech Disentanglement","Signal Processing"]}]},{"id":"language_rights","label":"Language and Rights","entries":[{"key":"dc:language","label":"Dc Language","values":["en","eng"]},{"key":"dc:rights","label":"Dc Rights","values":["Copyright 2022 Chak Ho Chan"]}]},{"id":"identifiers","label":"Identifiers","entries":[{"key":"dc:identifier","label":"Identifier","values":["https://hdl.handle.net/2142/117680"]}]},{"id":"additional","label":"Additional Metadata","entries":[{"key":"dc:description","label":"Description","values":["Submission published under a 24 month embargo labeled 'U of I Access', the embargo will last until 2024-12-01","The student, Chak Ho Chan, accepted the attached license on 2022-12-01 at 13:22.","The student, Chak Ho Chan, submitted this Thesis for approval on 2022-12-01 at 13:53.","This Thesis was approved for publication on 2022-12-07 at 08:41.","DSpace SAF Submission Ingestion Package generated from Vireo submission #18719 on 2023-04-12 at 08:13:30","SpeechSplit is among the first algorithms that successfully disentangle speech into four components: rhythm, content, pitch, and timbre. However, the model requires exhaustive tuning of the encoder bottlenecks, which can be a daunting task and limits its generalization ability. In this work, we present SpeechSplit2, an improved version of SpeechSplit, in which simple signal processing methods are utilized to alleviate the laborious bottleneck tuning problem. We show that by feeding different inputs to each encoder, we can guide each encoder to only extract one particular aspect of speech and discard the rest, given the bottleneck size is sufficiently large to encode the corresponding information. With the same neural network architecture as SpeechSplit, SpeechSplit2 achieves comparable performance in disentangling speech components when the bottlenecks are carefully tuned and shows superior advantage over the baseline when the bottleneck size varies."]},{"key":"dc:format","label":"Dc Format","values":["application/pdf"]},{"key":"dc:title","label":"Title","values":["SpeechSplit2: An efficient unsupervised speech disentanglement model for multi-aspect voice conversion"]}]}],"canonical_facts":{"dc:contributor":["Hasegawa-Johnson, Mark Allan"],"dc:creator":["Chan, Chak Ho"],"dc:date":["2022-12","2022-12-07"],"dc:description":["Submission published under a 24 month embargo labeled 'U of I Access', the embargo will last until 2024-12-01","The student, Chak Ho Chan, accepted the attached license on 2022-12-01 at 13:22.","The student, Chak Ho Chan, submitted this Thesis for approval on 2022-12-01 at 13:53.","This Thesis was approved for publication on 2022-12-07 at 08:41.","DSpace SAF Submission Ingestion Package generated from Vireo submission #18719 on 2023-04-12 at 08:13:30","SpeechSplit is among the first algorithms that successfully disentangle speech into four components: rhythm, content, pitch, and timbre. However, the model requires exhaustive tuning of the encoder bottlenecks, which can be a daunting task and limits its generalization ability. In this work, we present SpeechSplit2, an improved version of SpeechSplit, in which simple signal processing methods are utilized to alleviate the laborious bottleneck tuning problem. We show that by feeding different inputs to each encoder, we can guide each encoder to only extract one particular aspect of speech and discard the rest, given the bottleneck size is sufficiently large to encode the corresponding information. With the same neural network architecture as SpeechSplit, SpeechSplit2 achieves comparable performance in disentangling speech components when the bottlenecks are carefully tuned and shows superior advantage over the baseline when the bottleneck size varies."],"dc:format":["application/pdf"],"dc:identifier":["https://hdl.handle.net/2142/117680"],"dc:language":["en","eng"],"dc:rights":["Copyright 2022 Chak Ho Chan"],"dc:subject":["Unsupervised Learning","Voice Conversion","Speech Disentanglement","Signal Processing"],"dc:title":["SpeechSplit2: An efficient unsupervised speech disentanglement model for multi-aspect voice conversion"],"dc:type":["text","Thesis"],"thesis:degree_discipline":["Electrical & Computer Engr"],"thesis:degree_level":["Thesis"],"thesis:degree_name":["M.S."],"thesis:institution_name":["University of Illinois at Urbana-Champaign"]},"updated_at":"2026-07-22T22:24:56Z"}