{"id":{"repo_id":"uoit","oai_identifier":"oai:ontariotechu.scholaris.ca:10155/1358"},"canonical_url":"https://search.dev.ndltd.org/etd/uoit/oai:ontariotechu.scholaris.ca:10155/1358","repository":{"repo_id":"uoit","name":"Ontario Institute of Technology","base_url":"https://ontariotechu.scholaris.ca/server/oai/request"},"display":{"title":"SegmentPerturb: effective black-box hidden voice attack on commercial ASR systems via selective deletion","abstract":"Voice control systems continue becoming more pervasive as they are deployed in mobile phones, smart home devices, automobiles, etc. Commonly, voice control systems have high privileges on the device, such as making a call or placing an order. However, at the same time, they are vulnerable to voice attacks, which may lead to serious consequences. In this thesis, SegmentPerturb was proposed to craft hidden voice commands via inquiring the target models. The basic idea of SegmentPerturb is that the original command audio was separated into multiple segments and a certain degree of perturbation was applied to each segment by probing the target speech recognition system. Experiments were conducted on four popular commercial speech recognition APIs plus one smart home device to show the practicability of SegmentPerturb. Results suggest that SegmentPerturb can generate voice commands which can be recognized by the machine but are hard to understand by a human.","abstract_html":"Voice control systems continue becoming more pervasive as they are deployed in mobile phones, smart home devices, automobiles, etc. Commonly, voice control systems have high privileges on the device, such as making a call or placing an order. However, at the same time, they are vulnerable to voice attacks, which may lead to serious consequences. In this thesis, SegmentPerturb was proposed to craft hidden voice commands via inquiring the target models. The basic idea of SegmentPerturb is that the original command audio was separated into multiple segments and a certain degree of perturbation was applied to each segment by probing the target speech recognition system. Experiments were conducted on four popular commercial speech recognition APIs plus one smart home device to show the practicability of SegmentPerturb. Results suggest that SegmentPerturb can generate voice commands which can be recognized by the machine but are hard to understand by a human.","abstract_has_math":false,"creators":["Wang, Ganyu"],"institution":"University of Ontario Institute of Technology","degree_name":"Master of Science (MSc)","degree_level":null,"degree_discipline":"Computer Science","degree_department":null,"school":null,"contributors":[],"advisors":["Vargas Martin, Miguel"],"committee_chairs":[],"committee_members":[],"year":2021,"date_issued":"2021-08-01","date_published":"2021-08-01","updated_at":"2026-07-24T05:35:34Z","subjects":["Automatic Speech Recognition System","Hidden Voice Command"],"languages":["en"],"rights":[],"rights_urls":[],"identifier_entries":[]},"links":{"outbound_url":"https://hdl.handle.net/10155/1358","outbound_label":"Handle","outbound_source":"dc:identifier.uri"},"metadata_groups":[{"id":"people","label":"People","entries":[{"key":"dc:contributor.advisor","label":"Advisor","values":["Vargas Martin, Miguel"]},{"key":"dc:creator","label":"Author","values":["Wang, Ganyu"]}]},{"id":"academic_context","label":"Academic Context","entries":[{"key":"dc:date.accessioned","label":"Dc Date Accessioned","values":["2021-10-01T19:43:23Z","2022-03-29T17:27:03Z"]},{"key":"dc:date.available","label":"Dc Date Available","values":["2021-10-01T19:43:23Z","2022-03-29T17:27:03Z"]},{"key":"dc:date.issued","label":"Date","values":["2021-08-01"]},{"key":"dc:type","label":"Dc Type","values":["Thesis"]},{"key":"thesis:degree_discipline","label":"Discipline","values":["Computer Science"]},{"key":"thesis:degree_name","label":"Degree Name","values":["Master of Science (MSc)"]},{"key":"thesis:institution_name","label":"Thesis Institution Name","values":["University of Ontario Institute of Technology"]}]},{"id":"subjects_keywords","label":"Subjects and Keywords","entries":[{"key":"dc:subject","label":"Dc Subject","values":["Automatic Speech Recognition System","Hidden Voice Command"]}]},{"id":"language_rights","label":"Language and Rights","entries":[{"key":"dc:language.iso","label":"Language (ISO)","values":["en"]}]},{"id":"identifiers","label":"Identifiers","entries":[{"key":"dc:identifier.uri","label":"Identifier URI","values":["https://hdl.handle.net/10155/1358"]}]},{"id":"additional","label":"Additional Metadata","entries":[{"key":"dc:description.abstract","label":"Abstract","values":["Voice control systems continue becoming more pervasive as they are deployed in mobile phones, smart home devices, automobiles, etc. Commonly, voice control systems have high privileges on the device, such as making a call or placing an order. However, at the same time, they are vulnerable to voice attacks, which may lead to serious consequences. In this thesis, SegmentPerturb was proposed to craft hidden voice commands via inquiring the target models. The basic idea of SegmentPerturb is that the original command audio was separated into multiple segments and a certain degree of perturbation was applied to each segment by probing the target speech recognition system. Experiments were conducted on four popular commercial speech recognition APIs plus one smart home device to show the practicability of SegmentPerturb. Results suggest that SegmentPerturb can generate voice commands which can be recognized by the machine but are hard to understand by a human."]},{"key":"dc:title","label":"Title","values":["SegmentPerturb: effective black-box hidden voice attack on commercial ASR systems via selective deletion"]}]}],"canonical_facts":{"dc:contributor.advisor":["Vargas Martin, Miguel"],"dc:creator":["Wang, Ganyu"],"dc:date.accessioned":["2021-10-01T19:43:23Z","2022-03-29T17:27:03Z"],"dc:date.available":["2021-10-01T19:43:23Z","2022-03-29T17:27:03Z"],"dc:date.issued":["2021-08-01"],"dc:description.abstract":["Voice control systems continue becoming more pervasive as they are deployed in mobile phones, smart home devices, automobiles, etc. Commonly, voice control systems have high privileges on the device, such as making a call or placing an order. However, at the same time, they are vulnerable to voice attacks, which may lead to serious consequences. In this thesis, SegmentPerturb was proposed to craft hidden voice commands via inquiring the target models. The basic idea of SegmentPerturb is that the original command audio was separated into multiple segments and a certain degree of perturbation was applied to each segment by probing the target speech recognition system. Experiments were conducted on four popular commercial speech recognition APIs plus one smart home device to show the practicability of SegmentPerturb. Results suggest that SegmentPerturb can generate voice commands which can be recognized by the machine but are hard to understand by a human."],"dc:identifier.uri":["https://hdl.handle.net/10155/1358"],"dc:language.iso":["en"],"dc:subject":["Automatic Speech Recognition System","Hidden Voice Command"],"dc:title":["SegmentPerturb: effective black-box hidden voice attack on commercial ASR systems via selective deletion"],"dc:type":["Thesis"],"thesis:degree_discipline":["Computer Science"],"thesis:degree_name":["Master of Science (MSc)"],"thesis:institution_name":["University of Ontario Institute of Technology"]},"updated_at":"2026-07-24T05:35:34Z"}