{"id":{"repo_id":"uiuc","oai_identifier":"oai:www.ideals.illinois.edu:2142/18304"},"canonical_url":"https://search.dev.ndltd.org/etd/uiuc/oai:www.ideals.illinois.edu:2142/18304","repository":{"repo_id":"uiuc","name":"University of Illinois - Urbana-Champaign","base_url":"https://www.ideals.illinois.edu/oai-pmh"},"display":{"title":"Data Cleaning Framework: An Extensible Approach to Data Cleaning","abstract":"The growing dependence of society on enormous quantities of information stored electronically has led to a corresponding rise in errors in this information. The stored data can be critically important, necessitating new ways of correcting anomalous records. Current cleaning techniques are very domain-specific and hard to extend, hindering their use in some areas. This work proposes an extensible framework for data cleaning, allowing users to customize the cleaning to their specific requirements. It defines categories of common cleaning operations, allowing more robust support for user-implemented cleaning functions in these categories. The experimental results show that the proposed data cleaning framework is an effective approach to cleaning data for arbitrary domains.","abstract_html":"The growing dependence of society on enormous quantities of information stored electronically has led to a corresponding rise in errors in this information. The stored data can be critically important, necessitating new ways of correcting anomalous records. Current cleaning techniques are very domain-specific and hard to extend, hindering their use in some areas. This work proposes an extensible framework for data cleaning, allowing users to customize the cleaning to their specific requirements. It defines categories of common cleaning operations, allowing more robust support for user-implemented cleaning functions in these categories. The experimental results show that the proposed data cleaning framework is an effective approach to cleaning data for arbitrary domains.","abstract_has_math":false,"creators":["Gu, Randy S."],"institution":"University of Illinois at Urbana-Champaign","degree_name":"M.S.","degree_level":"Thesis","degree_discipline":"Computer Science","degree_department":null,"school":null,"contributors":["Chang, Kevin C-C."],"advisors":[],"committee_chairs":[],"committee_members":[],"year":2011,"date_issued":"2011-01-14T22:45:32Z","date_published":"2011-01-14T22:45:32Z","updated_at":"2026-07-22T22:25:11Z","subjects":["Data Cleaning"],"languages":["en"],"rights":["Copyright 2010 Randy Siran Gu"],"rights_urls":[],"identifier_entries":[]},"links":{"outbound_url":"http://hdl.handle.net/2142/18304","outbound_label":"Handle","outbound_source":"dc:identifier"},"metadata_groups":[{"id":"people","label":"People","entries":[{"key":"dc:contributor","label":"Contributor","values":["Chang, Kevin C-C."]},{"key":"dc:creator","label":"Author","values":["Gu, Randy S."]}]},{"id":"academic_context","label":"Academic Context","entries":[{"key":"dc:date","label":"Dc Date","values":["2011-01-14T22:45:32Z","2010-12"]},{"key":"thesis:degree_discipline","label":"Discipline","values":["Computer Science"]},{"key":"thesis:degree_level","label":"Degree Level","values":["Thesis"]},{"key":"thesis:degree_name","label":"Degree Name","values":["M.S."]},{"key":"thesis:institution_name","label":"Thesis Institution Name","values":["University of Illinois at Urbana-Champaign"]}]},{"id":"subjects_keywords","label":"Subjects and Keywords","entries":[{"key":"dc:subject","label":"Dc Subject","values":["Data Cleaning"]}]},{"id":"language_rights","label":"Language and Rights","entries":[{"key":"dc:language","label":"Dc Language","values":["en"]},{"key":"dc:rights","label":"Dc Rights","values":["Copyright 2010 Randy Siran Gu"]}]},{"id":"identifiers","label":"Identifiers","entries":[{"key":"dc:identifier","label":"Identifier","values":["http://hdl.handle.net/2142/18304"]}]},{"id":"additional","label":"Additional Metadata","entries":[{"key":"dc:description","label":"Description","values":["The growing dependence of society on enormous quantities of information stored electronically has led to a corresponding rise in errors in this information. The stored data can be critically important, necessitating new ways of correcting anomalous records. Current cleaning techniques are very domain-specific and hard to extend, hindering their use in some areas. This work proposes an extensible framework for data cleaning, allowing users to customize the cleaning to their specific requirements. It defines categories of common cleaning operations, allowing more robust support for user-implemented cleaning functions in these categories. The experimental results show that the proposed data cleaning framework is an effective approach to cleaning data for arbitrary domains.","Item withdrawn by Mark Zulauf (zulauf@illinois.edu) on 2010-12-06T22:32:50Z Item was in collections: University of Illinois Theses & Dissertations (ID: 1) No. of bitstreams: 1 Gu_Randy.pdf: 1018675 bytes, checksum: 99083517d8b9a0f46938d0a6b33c1284 (MD5)","Made available in DSpace on 2011-01-14T22:45:32Z (GMT). No. of bitstreams: 2 Gu_Randy.pdf: 1019060 bytes, checksum: a6e102659d31da4775e3eb22ede8e1e4 (MD5) license.txt: 4058 bytes, checksum: 03e2936f9233e17d6665b7c37e1b14ed (MD5)"]},{"key":"dc:title","label":"Title","values":["Data Cleaning Framework: An Extensible Approach to Data Cleaning"]}]}],"canonical_facts":{"dc:contributor":["Chang, Kevin C-C."],"dc:creator":["Gu, Randy S."],"dc:date":["2011-01-14T22:45:32Z","2010-12"],"dc:description":["The growing dependence of society on enormous quantities of information stored electronically has led to a corresponding rise in errors in this information. The stored data can be critically important, necessitating new ways of correcting anomalous records. Current cleaning techniques are very domain-specific and hard to extend, hindering their use in some areas. This work proposes an extensible framework for data cleaning, allowing users to customize the cleaning to their specific requirements. It defines categories of common cleaning operations, allowing more robust support for user-implemented cleaning functions in these categories. The experimental results show that the proposed data cleaning framework is an effective approach to cleaning data for arbitrary domains.","Item withdrawn by Mark Zulauf (zulauf@illinois.edu) on 2010-12-06T22:32:50Z Item was in collections: University of Illinois Theses & Dissertations (ID: 1) No. of bitstreams: 1 Gu_Randy.pdf: 1018675 bytes, checksum: 99083517d8b9a0f46938d0a6b33c1284 (MD5)","Made available in DSpace on 2011-01-14T22:45:32Z (GMT). No. of bitstreams: 2 Gu_Randy.pdf: 1019060 bytes, checksum: a6e102659d31da4775e3eb22ede8e1e4 (MD5) license.txt: 4058 bytes, checksum: 03e2936f9233e17d6665b7c37e1b14ed (MD5)"],"dc:identifier":["http://hdl.handle.net/2142/18304"],"dc:language":["en"],"dc:rights":["Copyright 2010 Randy Siran Gu"],"dc:subject":["Data Cleaning"],"dc:title":["Data Cleaning Framework: An Extensible Approach to Data Cleaning"],"thesis:degree_discipline":["Computer Science"],"thesis:degree_level":["Thesis"],"thesis:degree_name":["M.S."],"thesis:institution_name":["University of Illinois at Urbana-Champaign"]},"updated_at":"2026-07-22T22:25:11Z"}