{"id":{"repo_id":"brazil-uerj","oai_identifier":"oai:pantheon.ufrj.br:11422/6206"},"canonical_url":"https://search.dev.ndltd.org/etd/brazil-uerj/oai:pantheon.ufrj.br:11422/6206","repository":{"repo_id":"brazil-uerj","name":"Brazil UERJ","base_url":"https://pantheon.ufrj.br/oai/request"},"display":{"title":"Técnicas para conversão de orador em sinais de voz","abstract":"Presents a voice conversion system, a system that transforms a voice signal spoken by some speaker into a signal that sounds like it was spoken by another speaker, without changing the textual content of the speech or changing information like emotion or emphasis. The main objective of this work is to compare the conversion as done by different methods. To accomplish this, a unified voice conversion system containing the analysis, conversion and synthesis steps necessary to transform the speaker was implemented. Four voice conversion techniques, three from the literature, based on Gaussian mixture models, hidden Markov models and feed forward neural networks, and one novel based on recurrent neural networks, were evaluated. Two methods to generate the excitation used in the synthesis step were also implemented, one utilizing a parametric pulse trained on the speech signals, and one utilizing the PSOLA algorithm. On this system a couple of experiments were conducted to assess the conversion quality of each method: one measuring the distance between the cepstra of the signals, and the other employing a speaker recognition system. In these experiments the conversion based on Gaussian mixture models yielded the best results, but all techniques were relatively close in terms of performance.","abstract_html":"Presents a voice conversion system, a system that transforms a voice signal spoken by some speaker into a signal that sounds like it was spoken by another speaker, without changing the textual content of the speech or changing information like emotion or emphasis. The main objective of this work is to compare the conversion as done by different methods. To accomplish this, a unified voice conversion system containing the analysis, conversion and synthesis steps necessary to transform the speaker was implemented. Four voice conversion techniques, three from the literature, based on Gaussian mixture models, hidden Markov models and feed forward neural networks, and one novel based on recurrent neural networks, were evaluated. Two methods to generate the excitation used in the synthesis step were also implemented, one utilizing a parametric pulse trained on the speech signals, and one utilizing the PSOLA algorithm. On this system a couple of experiments were conducted to assess the conversion quality of each method: one measuring the distance between the cepstra of the signals, and the other employing a speaker recognition system. In these experiments the conversion based on Gaussian mixture models yielded the best results, but all techniques were relatively close in terms of performance.","abstract_has_math":false,"creators":["Costa, Victor Pereira da"],"institution":"Universidade Federal do Rio de Janeiro","degree_name":null,"degree_level":null,"degree_discipline":null,"degree_department":null,"school":null,"contributors":[],"advisors":["Biscainho, Luiz Wagner Pereira"],"committee_chairs":[],"committee_members":[],"year":2017,"date_issued":"2017-03","date_published":"2017-03","updated_at":"2026-07-24T01:16:18Z","subjects":["Processamento digital de voz","Processamento de sinais","Reconhecimento de voz"],"languages":["por"],"rights":["Acesso Aberto"],"rights_urls":[],"identifier_entries":[]},"links":{"outbound_url":"http://hdl.handle.net/11422/6206","outbound_label":"Handle","outbound_source":"dc:identifier.uri"},"metadata_groups":[{"id":"people","label":"People","entries":[{"key":"dc:contributor.advisor","label":"Advisor","values":["Biscainho, Luiz Wagner Pereira"]},{"key":"dc:creator","label":"Author","values":["Costa, Victor Pereira da"]}]},{"id":"academic_context","label":"Academic Context","entries":[{"key":"dc:date.accessioned","label":"Dc Date Accessioned","values":["2019-01-22T13:29:55Z"]},{"key":"dc:date.available","label":"Dc Date Available","values":["2026-05-16T03:05:35Z"]},{"key":"dc:date.issued","label":"Date","values":["2017-03"]},{"key":"dc:publisher","label":"Institution","values":["Universidade Federal do Rio de Janeiro"]},{"key":"dc:publisher.department","label":"Dc Publisher Department","values":["Instituto Alberto Luiz Coimbra de Pós-Graduação e Pesquisa de Engenharia"]},{"key":"dc:type","label":"Dc Type","values":["Dissertação"]}]},{"id":"subjects_keywords","label":"Subjects and Keywords","entries":[{"key":"dc:subject","label":"Dc Subject","values":["Processamento digital de voz","Processamento de sinais","Reconhecimento de voz"]}]},{"id":"language_rights","label":"Language and Rights","entries":[{"key":"dc:language","label":"Dc Language","values":["por"]},{"key":"dc:rights","label":"Dc Rights","values":["Acesso Aberto"]}]},{"id":"identifiers","label":"Identifiers","entries":[{"key":"dc:identifier.uri","label":"Identifier URI","values":["http://hdl.handle.net/11422/6206"]}]},{"id":"additional","label":"Additional Metadata","entries":[{"key":"dc:description.abstract","label":"Abstract","values":["Presents a voice conversion system, a system that transforms a voice signal spoken by some speaker into a signal that sounds like it was spoken by another speaker, without changing the textual content of the speech or changing information like emotion or emphasis. The main objective of this work is to compare the conversion as done by different methods. To accomplish this, a unified voice conversion system containing the analysis, conversion and synthesis steps necessary to transform the speaker was implemented. Four voice conversion techniques, three from the literature, based on Gaussian mixture models, hidden Markov models and feed forward neural networks, and one novel based on recurrent neural networks, were evaluated. Two methods to generate the excitation used in the synthesis step were also implemented, one utilizing a parametric pulse trained on the speech signals, and one utilizing the PSOLA algorithm. On this system a couple of experiments were conducted to assess the conversion quality of each method: one measuring the distance between the cepstra of the signals, and the other employing a speaker recognition system. In these experiments the conversion based on Gaussian mixture models yielded the best results, but all techniques were relatively close in terms of performance."]},{"key":"dc:title","label":"Title","values":["Técnicas para conversão de orador em sinais de voz"]}]}],"canonical_facts":{"dc:contributor.advisor":["Biscainho, Luiz Wagner Pereira"],"dc:creator":["Costa, Victor Pereira da"],"dc:date.accessioned":["2019-01-22T13:29:55Z"],"dc:date.available":["2026-05-16T03:05:35Z"],"dc:date.issued":["2017-03"],"dc:description.abstract":["Presents a voice conversion system, a system that transforms a voice signal spoken by some speaker into a signal that sounds like it was spoken by another speaker, without changing the textual content of the speech or changing information like emotion or emphasis. The main objective of this work is to compare the conversion as done by different methods. To accomplish this, a unified voice conversion system containing the analysis, conversion and synthesis steps necessary to transform the speaker was implemented. Four voice conversion techniques, three from the literature, based on Gaussian mixture models, hidden Markov models and feed forward neural networks, and one novel based on recurrent neural networks, were evaluated. Two methods to generate the excitation used in the synthesis step were also implemented, one utilizing a parametric pulse trained on the speech signals, and one utilizing the PSOLA algorithm. On this system a couple of experiments were conducted to assess the conversion quality of each method: one measuring the distance between the cepstra of the signals, and the other employing a speaker recognition system. In these experiments the conversion based on Gaussian mixture models yielded the best results, but all techniques were relatively close in terms of performance."],"dc:identifier.uri":["http://hdl.handle.net/11422/6206"],"dc:language":["por"],"dc:publisher":["Universidade Federal do Rio de Janeiro"],"dc:publisher.department":["Instituto Alberto Luiz Coimbra de Pós-Graduação e Pesquisa de Engenharia"],"dc:rights":["Acesso Aberto"],"dc:subject":["Processamento digital de voz","Processamento de sinais","Reconhecimento de voz"],"dc:title":["Técnicas para conversão de orador em sinais de voz"],"dc:type":["Dissertação"]},"updated_at":"2026-07-24T01:16:18Z"}