{"id":{"repo_id":"uiuc","oai_identifier":"oai:www.ideals.illinois.edu:2142/81543"},"canonical_url":"https://search.dev.ndltd.org/etd/uiuc/oai:www.ideals.illinois.edu:2142/81543","repository":{"repo_id":"uiuc","name":"University of Illinois - Urbana-Champaign","base_url":"https://www.ideals.illinois.edu/oai-pmh"},"display":{"title":"An Evaluation of Text Classification Methods for Literary Study","abstract":"Some of our conclusions are consistent with what are obtained in topic classification, such as Odds Ratio does not improve SVM performance and stop word removal might harm classification. Some conclusions contradict previous results, such as SVM does not beat naive Bayes in both cases. Some findings are new to this area---SVM and naive Bayes select top features in different frequency ranges; stemming might harm feature selection methods. These experiment results provide new insights to the relation between classification methods, feature engineering options and non-topic document properties. They also provide guidance for classification method selection in literary text classification applications.","abstract_html":"Some of our conclusions are consistent with what are obtained in topic classification, such as Odds Ratio does not improve SVM performance and stop word removal might harm classification. Some conclusions contradict previous results, such as SVM does not beat naive Bayes in both cases. Some findings are new to this area---SVM and naive Bayes select top features in different frequency ranges; stemming might harm feature selection methods. These experiment results provide new insights to the relation between classification methods, feature engineering options and non-topic document properties. They also provide guidance for classification method selection in literary text classification applications.","abstract_has_math":false,"creators":["Yu, Bei"],"institution":"University of Illinois at Urbana-Champaign","degree_name":"Ph.D.","degree_level":"Dissertation","degree_discipline":"Library and Information Science","degree_department":null,"school":null,"contributors":["Linda Smith"],"advisors":[],"committee_chairs":[],"committee_members":[],"year":2015,"date_issued":"2015-09-25T20:17:10Z","date_published":"2015-09-25T20:17:10Z","updated_at":"2026-07-22T22:26:16Z","subjects":["Literature, American"],"languages":["eng"],"rights":[],"rights_urls":[],"identifier_entries":[{"key":"dc:identifier","label":"Identifier","values":["(MiAaPQ)AAI3250350"],"render_values":[{"text":"(MiAaPQ)AAI3250350","href":null,"code":true}]}]},"links":{"outbound_url":"http://hdl.handle.net/2142/81543","outbound_label":"Handle","outbound_source":"dc:identifier"},"metadata_groups":[{"id":"people","label":"People","entries":[{"key":"dc:contributor","label":"Contributor","values":["Linda Smith"]},{"key":"dc:creator","label":"Author","values":["Yu, Bei"]}]},{"id":"academic_context","label":"Academic Context","entries":[{"key":"dc:date","label":"Dc Date","values":["2015-09-25T20:17:10Z","10000-01-01","2006"]},{"key":"dc:type","label":"Dc Type","values":["text"]},{"key":"thesis:degree_discipline","label":"Discipline","values":["Library and Information Science"]},{"key":"thesis:degree_level","label":"Degree Level","values":["Dissertation"]},{"key":"thesis:degree_name","label":"Degree Name","values":["Ph.D."]},{"key":"thesis:institution_name","label":"Thesis Institution Name","values":["University of Illinois at Urbana-Champaign"]}]},{"id":"subjects_keywords","label":"Subjects and Keywords","entries":[{"key":"dc:subject","label":"Dc Subject","values":["Literature, American"]}]},{"id":"language_rights","label":"Language and Rights","entries":[{"key":"dc:language","label":"Dc Language","values":["eng"]}]},{"id":"identifiers","label":"Identifiers","entries":[{"key":"dc:identifier","label":"Identifier","values":["http://hdl.handle.net/2142/81543","(MiAaPQ)AAI3250350"]}]},{"id":"additional","label":"Additional Metadata","entries":[{"key":"dc:description","label":"Description","values":["Some of our conclusions are consistent with what are obtained in topic classification, such as Odds Ratio does not improve SVM performance and stop word removal might harm classification. Some conclusions contradict previous results, such as SVM does not beat naive Bayes in both cases. Some findings are new to this area---SVM and naive Bayes select top features in different frequency ranges; stemming might harm feature selection methods. These experiment results provide new insights to the relation between classification methods, feature engineering options and non-topic document properties. They also provide guidance for classification method selection in literary text classification applications.","Made available in DSpace on 2015-09-25T20:17:10Z (GMT). No. of bitstreams: 2 license.txt: 4848 bytes, checksum: 96035ab3f5e1c23cc7138a224ce498bd (MD5) 3250350.pdf: 2849282 bytes, checksum: 4b8f4f71692b2809a97621757dacd4cf (MD5) Previous issue date: 2006","Embargo set by: Seth Robbins for item 82824 Lift date: Forever Reason: Restricted to the U of I community idenfinitely during batch ingest of legacy ETDs","Restricted to the U of I community idenfinitely during batch ingest of legacy ETDs","U of I Only","116 p.","Thesis (Ph.D.)--University of Illinois at Urbana-Champaign, 2006."]},{"key":"dc:title","label":"Title","values":["An Evaluation of Text Classification Methods for Literary Study"]}]}],"canonical_facts":{"dc:contributor":["Linda Smith"],"dc:creator":["Yu, Bei"],"dc:date":["2015-09-25T20:17:10Z","10000-01-01","2006"],"dc:description":["Some of our conclusions are consistent with what are obtained in topic classification, such as Odds Ratio does not improve SVM performance and stop word removal might harm classification. Some conclusions contradict previous results, such as SVM does not beat naive Bayes in both cases. Some findings are new to this area---SVM and naive Bayes select top features in different frequency ranges; stemming might harm feature selection methods. These experiment results provide new insights to the relation between classification methods, feature engineering options and non-topic document properties. They also provide guidance for classification method selection in literary text classification applications.","Made available in DSpace on 2015-09-25T20:17:10Z (GMT). No. of bitstreams: 2 license.txt: 4848 bytes, checksum: 96035ab3f5e1c23cc7138a224ce498bd (MD5) 3250350.pdf: 2849282 bytes, checksum: 4b8f4f71692b2809a97621757dacd4cf (MD5) Previous issue date: 2006","Embargo set by: Seth Robbins for item 82824 Lift date: Forever Reason: Restricted to the U of I community idenfinitely during batch ingest of legacy ETDs","Restricted to the U of I community idenfinitely during batch ingest of legacy ETDs","U of I Only","116 p.","Thesis (Ph.D.)--University of Illinois at Urbana-Champaign, 2006."],"dc:identifier":["http://hdl.handle.net/2142/81543","(MiAaPQ)AAI3250350"],"dc:language":["eng"],"dc:subject":["Literature, American"],"dc:title":["An Evaluation of Text Classification Methods for Literary Study"],"dc:type":["text"],"thesis:degree_discipline":["Library and Information Science"],"thesis:degree_level":["Dissertation"],"thesis:degree_name":["Ph.D."],"thesis:institution_name":["University of Illinois at Urbana-Champaign"]},"updated_at":"2026-07-22T22:26:16Z"}