{"id":{"repo_id":"birmingham","oai_identifier":"oai:etheses.bham.ac.uk:116"},"canonical_url":"https://search.dev.ndltd.org/etd/birmingham/oai:etheses.bham.ac.uk:116","repository":{"repo_id":"birmingham","name":"University of Birmingham","base_url":"https://etheses.bham.ac.uk/cgi/oai2"},"display":{"title":"The automatic extraction of linguistic information from text corpora","abstract":"This is a study exploring the feasibility of a fully automated analysis of linguistic data. It identifies a requirement for large-scale investigations, which cannot be done manually by a human researcher. Instead, methods from natural language processing are suggested as a way to analyse large amounts of corpus data without any human intervention. Human involvement hinders scalability and introduces a bias which prevents studies from being completely replicable. The fundamental assumption underlying this work is that linguistic analysis must be empirical, and that reliance on existing theories or even descriptive categories should be avoided as far as possible. In this thesis we report the results of a number of case studies investigating various areas of language description, lexis, grammar, and meaning. The aim of these case studies is to see how far we can automate the analysis of different aspects of language, both with data gathering and subsequent processing of the data. The outcomes of the feasibility studies demonstrate the practicability of such automated analyses.","abstract_html":"This is a study exploring the feasibility of a fully automated analysis of linguistic data. It identifies a requirement for large-scale investigations, which cannot be done manually by a human researcher. Instead, methods from natural language processing are suggested as a way to analyse large amounts of corpus data without any human intervention. Human involvement hinders scalability and introduces a bias which prevents studies from being completely replicable. The fundamental assumption underlying this work is that linguistic analysis must be empirical, and that reliance on existing theories or even descriptive categories should be avoided as far as possible. In this thesis we report the results of a number of case studies investigating various areas of language description, lexis, grammar, and meaning. The aim of these case studies is to see how far we can automate the analysis of different aspects of language, both with data gathering and subsequent processing of the data. The outcomes of the feasibility studies demonstrate the practicability of such automated analyses.","abstract_has_math":false,"creators":["Mason, Oliver Jan"],"institution":"University of Birmingham","degree_name":"d_ph","degree_level":"d_ph","degree_discipline":null,"degree_department":null,"school":null,"contributors":[],"advisors":[],"committee_chairs":[],"committee_members":[],"year":2006,"date_issued":"2006-12","date_published":"2006-12","updated_at":"2026-07-24T01:10:46Z","subjects":["PE English","P Philology. Linguistics","QA76 Computer software"],"languages":[],"rights":[],"rights_urls":[],"identifier_entries":[]},"links":{"outbound_url":null,"outbound_label":null,"outbound_source":null},"metadata_groups":[{"id":"people","label":"People","entries":[{"key":"dc:contributor.sponsor","label":"Sponsor","values":["na"]},{"key":"dc:creator","label":"Author","values":["Mason, Oliver Jan"]}]},{"id":"academic_context","label":"Academic Context","entries":[{"key":"dc:date","label":"Dc Date","values":["2006-12"]},{"key":"dc:date.issued","label":"Date","values":["2006-12"]},{"key":"dc:publisher.department","label":"Dc Publisher Department","values":["School of Humanities","School of English, Drama and American & Canadian Studies, Department of English Literature"]},{"key":"dc:publisher.institution","label":"Dc Publisher Institution","values":["University of Birmingham"]},{"key":"dc:relation.isreferencedby","label":"Dc Relation Isreferencedby","values":["http://etheses.bham.ac.uk//id/eprint/116/"]},{"key":"dc:type","label":"Dc Type","values":["Thesis"]},{"key":"dc:type.qualificationlevel","label":"Dc Type Qualificationlevel","values":["d_ph"]},{"key":"dc:type.qualificationname","label":"Dc Type Qualificationname","values":["d_ph"]}]},{"id":"subjects_keywords","label":"Subjects and Keywords","entries":[{"key":"dc:subject","label":"Dc Subject","values":["PE English","P Philology. Linguistics","QA76 Computer software"]}]},{"id":"identifiers","label":"Identifiers","entries":[{"key":"dc:identifier.uri","label":"Identifier URI","values":["http://etheses.bham.ac.uk//id/eprint/116/1/Mason06PhD.pdf","http://etheses.bham.ac.uk//id/eprint/116/2/Decl_IS_Mason06PhD.pdf"]}]},{"id":"additional","label":"Additional Metadata","entries":[{"key":"dc:description.abstract","label":"Abstract","values":["This is a study exploring the feasibility of a fully automated analysis of linguistic data. It identifies a requirement for large-scale investigations, which cannot be done manually by a human researcher. Instead, methods from natural language processing are suggested as a way to analyse large amounts of corpus data without any human intervention. Human involvement hinders scalability and introduces a bias which prevents studies from being completely replicable. The fundamental assumption underlying this work is that linguistic analysis must be empirical, and that reliance on existing theories or even descriptive categories should be avoided as far as possible. In this thesis we report the results of a number of case studies investigating various areas of language description, lexis, grammar, and meaning. The aim of these case studies is to see how far we can automate the analysis of different aspects of language, both with data gathering and subsequent processing of the data. The outcomes of the feasibility studies demonstrate the practicability of such automated analyses."]},{"key":"dc:format","label":"Dc Format","values":["application/pdf"]},{"key":"dc:title","label":"Title","values":["The automatic extraction of linguistic information from text corpora"]}]}],"canonical_facts":{"dc:contributor.sponsor":["na"],"dc:creator":["Mason, Oliver Jan"],"dc:date":["2006-12"],"dc:date.issued":["2006-12"],"dc:description.abstract":["This is a study exploring the feasibility of a fully automated analysis of linguistic data. It identifies a requirement for large-scale investigations, which cannot be done manually by a human researcher. Instead, methods from natural language processing are suggested as a way to analyse large amounts of corpus data without any human intervention. Human involvement hinders scalability and introduces a bias which prevents studies from being completely replicable. The fundamental assumption underlying this work is that linguistic analysis must be empirical, and that reliance on existing theories or even descriptive categories should be avoided as far as possible. In this thesis we report the results of a number of case studies investigating various areas of language description, lexis, grammar, and meaning. The aim of these case studies is to see how far we can automate the analysis of different aspects of language, both with data gathering and subsequent processing of the data. The outcomes of the feasibility studies demonstrate the practicability of such automated analyses."],"dc:format":["application/pdf"],"dc:identifier.uri":["http://etheses.bham.ac.uk//id/eprint/116/1/Mason06PhD.pdf","http://etheses.bham.ac.uk//id/eprint/116/2/Decl_IS_Mason06PhD.pdf"],"dc:publisher.department":["School of Humanities","School of English, Drama and American & Canadian Studies, Department of English Literature"],"dc:publisher.institution":["University of Birmingham"],"dc:relation.isreferencedby":["http://etheses.bham.ac.uk//id/eprint/116/"],"dc:subject":["PE English","P Philology. Linguistics","QA76 Computer software"],"dc:title":["The automatic extraction of linguistic information from text corpora"],"dc:type":["Thesis"],"dc:type.qualificationlevel":["d_ph"],"dc:type.qualificationname":["d_ph"]},"updated_at":"2026-07-24T01:10:46Z"}