{"id":{"repo_id":"radboud","oai_identifier":"oai:repository.ubn.ru.nl:2066/27415"},"canonical_url":"https://search.dev.ndltd.org/etd/radboud/oai:repository.ubn.ru.nl:2066/27415","repository":{"repo_id":"radboud","name":"Radboud University Nijmegen","base_url":"https://repository.ubn.ru.nl/oai/request"},"display":{"title":"Phonetic Transcriptions of Large Speech Corpora","abstract":"Contains fulltext : 27415.pdf (Publisher’s version ) (Open Access) Contains fulltext : 41404.pdf (Publisher’s version ) (Open Access)","abstract_html":"Contains fulltext : 27415.pdf (Publisher’s version ) (Open Access) Contains fulltext : 41404.pdf (Publisher’s version ) (Open Access)","abstract_has_math":false,"creators":["Binnenpoorte, D.M."],"institution":"S.l. : s.n.","degree_name":null,"degree_level":null,"degree_discipline":null,"degree_department":null,"school":null,"contributors":["Boves, L.W.J.","Cucchiarini, C."],"advisors":[],"committee_chairs":[],"committee_members":[],"year":2006,"date_issued":"2006","date_published":"2006","updated_at":"2026-07-24T04:02:38Z","subjects":["Automatic Phonetic Transcriptions","Linguistic Information Processing"],"languages":[],"rights":[],"rights_urls":[],"identifier_entries":[{"key":"dc:identifier","label":"Identifier","values":["https://hdl.handle.net/2066/41404"],"render_values":[{"text":"https://hdl.handle.net/2066/41404","href":"https://hdl.handle.net/2066/41404","code":true}]}]},"links":{"outbound_url":"https://hdl.handle.net/2066/27415","outbound_label":"Handle","outbound_source":"dc:identifier"},"metadata_groups":[{"id":"people","label":"People","entries":[{"key":"dc:contributor","label":"Contributor","values":["Boves, L.W.J.","Cucchiarini, C."]},{"key":"dc:creator","label":"Author","values":["Binnenpoorte, D.M."]}]},{"id":"academic_context","label":"Academic Context","entries":[{"key":"dc:date","label":"Dc Date","values":["2006"]},{"key":"dc:publisher","label":"Institution","values":["S.l. : s.n."]},{"key":"dc:type","label":"Dc Type","values":["Doctoral thesis"]}]},{"id":"subjects_keywords","label":"Subjects and Keywords","entries":[{"key":"dc:subject","label":"Dc Subject","values":["Automatic Phonetic Transcriptions","Linguistic Information Processing"]}]},{"id":"identifiers","label":"Identifiers","entries":[{"key":"dc:identifier","label":"Identifier","values":["https://repository.ubn.ru.nl//bitstream/handle/2066/27415/27415.pdf","https://hdl.handle.net/2066/27415","https://hdl.handle.net/2066/41404"]}]},{"id":"additional","label":"Additional Metadata","entries":[{"key":"dc:description","label":"Description","values":["Contains fulltext : 27415.pdf (Publisher’s version ) (Open Access) Contains fulltext : 41404.pdf (Publisher’s version ) (Open Access)","Each time a word is uttered, even pronounced by one and the same speaker, its pronunciation can differ, and may also be rather different from the canonical transcription. For research on pronunciation phenomena numerous samples of real-life speech need to be collected. Such large collections of speech are called speech corpora. Speech corpora constitute a rich resource for empirical investigations on spoken language. However, in order to be useful as a resource for pronunciation research, it is necessary to have phonetic transcriptions. The research reported on in this thesis is focused on, first, the generation of phonetic transcriptions of large speech corpora, and in relation to this, on gathering new phonological knowledge, and finally, on the evaluation of the quality of phonetic transcriptions. Since a complete manual phonetic transcription of a large speech corpus is practically impossible, recourse to automatic techniques is inevitable. We successfully developed and tested both data-driven and knowledge-based automatic phonetic transcription generation procedures that not only yielded more accurate transcriptions, but also increased the phonological knowledge on pronunciation phenomena in real-life speech (Dutch). Furthermore, a quality measure was developed for more objective assessments of both automatically and manually generated phonetic transcriptions. In general it can be concluded, that good quality broad phonetic transcription for read speech can be obtained fully automatically by using relatively simple techniques. By omitting human-made transcriptions of read speech, a lot of time and money can be saved that can be allocated for the benefit of phonetic transcriptions of speech styles for which larger deviations from a canonical representations are to be expected. For spontaneous speech human transcriptions are still the best option, although improved automatic techniques, together with a better understanding of the phonological processes underlying spontaneous speech, are likely to approximate human transcription quality for spontaneous speech styles in the future.","Radboud Universiteit Nijmegen, 07 april 2006","Promotor : Boves, L.W.J. Co-promotor : Cucchiarini, C.","X, 158 p."]},{"key":"dc:title","label":"Title","values":["Phonetic Transcriptions of Large Speech Corpora"]}]}],"canonical_facts":{"dc:contributor":["Boves, L.W.J.","Cucchiarini, C."],"dc:creator":["Binnenpoorte, D.M."],"dc:date":["2006"],"dc:description":["Contains fulltext : 27415.pdf (Publisher’s version ) (Open Access) Contains fulltext : 41404.pdf (Publisher’s version ) (Open Access)","Each time a word is uttered, even pronounced by one and the same speaker, its pronunciation can differ, and may also be rather different from the canonical transcription. For research on pronunciation phenomena numerous samples of real-life speech need to be collected. Such large collections of speech are called speech corpora. Speech corpora constitute a rich resource for empirical investigations on spoken language. However, in order to be useful as a resource for pronunciation research, it is necessary to have phonetic transcriptions. The research reported on in this thesis is focused on, first, the generation of phonetic transcriptions of large speech corpora, and in relation to this, on gathering new phonological knowledge, and finally, on the evaluation of the quality of phonetic transcriptions. Since a complete manual phonetic transcription of a large speech corpus is practically impossible, recourse to automatic techniques is inevitable. We successfully developed and tested both data-driven and knowledge-based automatic phonetic transcription generation procedures that not only yielded more accurate transcriptions, but also increased the phonological knowledge on pronunciation phenomena in real-life speech (Dutch). Furthermore, a quality measure was developed for more objective assessments of both automatically and manually generated phonetic transcriptions. In general it can be concluded, that good quality broad phonetic transcription for read speech can be obtained fully automatically by using relatively simple techniques. By omitting human-made transcriptions of read speech, a lot of time and money can be saved that can be allocated for the benefit of phonetic transcriptions of speech styles for which larger deviations from a canonical representations are to be expected. For spontaneous speech human transcriptions are still the best option, although improved automatic techniques, together with a better understanding of the phonological processes underlying spontaneous speech, are likely to approximate human transcription quality for spontaneous speech styles in the future.","Radboud Universiteit Nijmegen, 07 april 2006","Promotor : Boves, L.W.J. Co-promotor : Cucchiarini, C.","X, 158 p."],"dc:identifier":["https://repository.ubn.ru.nl//bitstream/handle/2066/27415/27415.pdf","https://hdl.handle.net/2066/27415","https://hdl.handle.net/2066/41404"],"dc:publisher":["S.l. : s.n."],"dc:subject":["Automatic Phonetic Transcriptions","Linguistic Information Processing"],"dc:title":["Phonetic Transcriptions of Large Speech Corpora"],"dc:type":["Doctoral thesis"]},"updated_at":"2026-07-24T04:02:38Z"}