{"id":{"repo_id":"iupui","oai_identifier":"oai:scholarworks.indianapolis.iu.edu:1805/3364"},"canonical_url":"https://search.dev.ndltd.org/etd/iupui/oai:scholarworks.indianapolis.iu.edu:1805/3364","repository":{"repo_id":"iupui","name":"IUPUI","base_url":"https://scholarworks.indianapolis.iu.edu/server/oai/request"},"display":{"title":"Query Segmentation For E-Commerce Sites","abstract":"Query segmentation module is an integral part of Natural Language Processing which analyzes users' query and divides them into separate phrases. Published works on the query segmentation focus on the web search using Google n-gram frequencies corpus or text retrieval from relational databases. However, this module is also useful in the domain of E-Commerce for product search. In this thesis, we will discuss query segmentation in the context of the E-Commerce area. We propose a hybrid unsupervised segmentation methodology which is based on prefix tree, mutual information and relative frequency count to compute the score of query pairs and involve Wikipedia for new words recognition. Furthermore, we use two unique E-Commerce evaluation methods to quantify the accuracy of our query segmentation method.","abstract_html":"Query segmentation module is an integral part of Natural Language Processing which analyzes users&#x27; query and divides them into separate phrases. Published works on the query segmentation focus on the web search using Google n-gram frequencies corpus or text retrieval from relational databases. However, this module is also useful in the domain of E-Commerce for product search. In this thesis, we will discuss query segmentation in the context of the E-Commerce area. We propose a hybrid unsupervised segmentation methodology which is based on prefix tree, mutual information and relative frequency count to compute the score of query pairs and involve Wikipedia for new words recognition. Furthermore, we use two unique E-Commerce evaluation methods to quantify the accuracy of our query segmentation method.","abstract_has_math":false,"creators":["Gong, Xiaojing"],"institution":null,"degree_name":null,"degree_level":null,"degree_discipline":null,"degree_department":null,"school":null,"contributors":[],"advisors":["Al Hasan, Mohammad"],"committee_chairs":[],"committee_members":[],"year":2012,"date_issued":"2012-08","date_published":"2012-08","updated_at":"2026-07-24T02:41:41Z","subjects":["Query Segmentation","prefix tree","unsupervised segmentation","E-Commerce"],"languages":["en_US"],"rights":[],"rights_urls":[],"identifier_entries":[{"key":"dc:identifier.uri","label":"Identifier URI","values":["http://dx.doi.org/10.7912/C2/2296"],"render_values":[{"text":"http://dx.doi.org/10.7912/C2/2296","href":"http://dx.doi.org/10.7912/C2/2296","code":true}]}]},"links":{"outbound_url":"https://hdl.handle.net/1805/3364","outbound_label":"Handle","outbound_source":"dc:identifier.uri"},"metadata_groups":[{"id":"people","label":"People","entries":[{"key":"dc:contributor.advisor","label":"Advisor","values":["Al Hasan, Mohammad"]},{"key":"dc:contributor.other","label":"Dc Contributor Other","values":["Fang, Shiaofen","Raje, Rajeev"]},{"key":"dc:creator","label":"Author","values":["Gong, Xiaojing"]}]},{"id":"academic_context","label":"Academic Context","entries":[{"key":"dc:date.accessioned","label":"Dc Date Accessioned","values":["2013-07-12T17:10:15Z"]},{"key":"dc:date.available","label":"Dc Date Available","values":["2013-07-12T17:10:15Z"]},{"key":"dc:date.issued","label":"Date","values":["2012-08"]},{"key":"dc:type","label":"Dc Type","values":["Thesis"]}]},{"id":"subjects_keywords","label":"Subjects and Keywords","entries":[{"key":"dc:subject","label":"Dc Subject","values":["Query Segmentation","prefix tree","unsupervised segmentation","E-Commerce"]}]},{"id":"language_rights","label":"Language and Rights","entries":[{"key":"dc:language.iso","label":"Language (ISO)","values":["en_US"]}]},{"id":"identifiers","label":"Identifiers","entries":[{"key":"dc:identifier.uri","label":"Identifier URI","values":["https://hdl.handle.net/1805/3364","http://dx.doi.org/10.7912/C2/2296"]}]},{"id":"additional","label":"Additional Metadata","entries":[{"key":"dc:description","label":"Description","values":["Indiana University-Purdue University Indianapolis (IUPUI)"]},{"key":"dc:description.abstract","label":"Abstract","values":["Query segmentation module is an integral part of Natural Language Processing which analyzes users' query and divides them into separate phrases. Published works on the query segmentation focus on the web search using Google n-gram frequencies corpus or text retrieval from relational databases. However, this module is also useful in the domain of E-Commerce for product search. In this thesis, we will discuss query segmentation in the context of the E-Commerce area. We propose a hybrid unsupervised segmentation methodology which is based on prefix tree, mutual information and relative frequency count to compute the score of query pairs and involve Wikipedia for new words recognition. Furthermore, we use two unique E-Commerce evaluation methods to quantify the accuracy of our query segmentation method."]},{"key":"dc:title","label":"Title","values":["Query Segmentation For E-Commerce Sites"]}]}],"canonical_facts":{"dc:contributor.advisor":["Al Hasan, Mohammad"],"dc:contributor.other":["Fang, Shiaofen","Raje, Rajeev"],"dc:creator":["Gong, Xiaojing"],"dc:date.accessioned":["2013-07-12T17:10:15Z"],"dc:date.available":["2013-07-12T17:10:15Z"],"dc:date.issued":["2012-08"],"dc:description":["Indiana University-Purdue University Indianapolis (IUPUI)"],"dc:description.abstract":["Query segmentation module is an integral part of Natural Language Processing which analyzes users' query and divides them into separate phrases. Published works on the query segmentation focus on the web search using Google n-gram frequencies corpus or text retrieval from relational databases. However, this module is also useful in the domain of E-Commerce for product search. In this thesis, we will discuss query segmentation in the context of the E-Commerce area. We propose a hybrid unsupervised segmentation methodology which is based on prefix tree, mutual information and relative frequency count to compute the score of query pairs and involve Wikipedia for new words recognition. Furthermore, we use two unique E-Commerce evaluation methods to quantify the accuracy of our query segmentation method."],"dc:identifier.uri":["https://hdl.handle.net/1805/3364","http://dx.doi.org/10.7912/C2/2296"],"dc:language.iso":["en_US"],"dc:subject":["Query Segmentation","prefix tree","unsupervised segmentation","E-Commerce"],"dc:title":["Query Segmentation For E-Commerce Sites"],"dc:type":["Thesis"]},"updated_at":"2026-07-24T02:41:41Z"}