{"id":{"repo_id":"uiuc","oai_identifier":"oai:www.ideals.illinois.edu:2142/120590"},"canonical_url":"https://search.dev.ndltd.org/etd/uiuc/oai:www.ideals.illinois.edu:2142/120590","repository":{"repo_id":"uiuc","name":"University of Illinois - Urbana-Champaign","base_url":"https://www.ideals.illinois.edu/oai-pmh"},"display":{"title":"User-guided dynamic topic discovery in large texts","abstract":"Submission published under a 24 month embargo labeled 'Closed Access', the embargo will last until 2025-05-01","abstract_html":"Submission published under a 24 month embargo labeled &#x27;Closed Access&#x27;, the embargo will last until 2025-05-01","abstract_has_math":false,"creators":["Venkat Ramanan, Karthik"],"institution":"University of Illinois at Urbana-Champaign","degree_name":"M.S.","degree_level":"Thesis","degree_discipline":"Computer Science","degree_department":null,"school":null,"contributors":["Han, Jiawei"],"advisors":[],"committee_chairs":[],"committee_members":[],"year":2023,"date_issued":"2023-05","date_published":"2023-05","updated_at":"2026-07-22T22:24:57Z","subjects":["Information Retrieval","Question Answering","Natural Language Processing","Data Mining","Pretrained Language Models","Topic Modeling"],"languages":["en","eng"],"rights":["Copyright 2023 Karthik Venkat Ramanan"],"rights_urls":[],"identifier_entries":[]},"links":{"outbound_url":"https://hdl.handle.net/2142/120590","outbound_label":"Handle","outbound_source":"dc:identifier"},"metadata_groups":[{"id":"people","label":"People","entries":[{"key":"dc:contributor","label":"Contributor","values":["Han, Jiawei"]},{"key":"dc:creator","label":"Author","values":["Venkat Ramanan, Karthik"]}]},{"id":"academic_context","label":"Academic Context","entries":[{"key":"dc:date","label":"Dc Date","values":["2023-05","2023-05-04"]},{"key":"dc:type","label":"Dc Type","values":["text"]},{"key":"thesis:degree_discipline","label":"Discipline","values":["Computer Science"]},{"key":"thesis:degree_level","label":"Degree Level","values":["Thesis"]},{"key":"thesis:degree_name","label":"Degree Name","values":["M.S."]},{"key":"thesis:institution_name","label":"Thesis Institution Name","values":["University of Illinois at Urbana-Champaign"]}]},{"id":"subjects_keywords","label":"Subjects and Keywords","entries":[{"key":"dc:subject","label":"Dc Subject","values":["Information Retrieval","Question Answering","Natural Language Processing","Data Mining","Pretrained Language Models","Topic Modeling"]}]},{"id":"language_rights","label":"Language and Rights","entries":[{"key":"dc:language","label":"Dc Language","values":["en","eng"]},{"key":"dc:rights","label":"Dc Rights","values":["Copyright 2023 Karthik Venkat Ramanan"]}]},{"id":"identifiers","label":"Identifiers","entries":[{"key":"dc:identifier","label":"Identifier","values":["https://hdl.handle.net/2142/120590"]}]},{"id":"additional","label":"Additional Metadata","entries":[{"key":"dc:description","label":"Description","values":["Submission published under a 24 month embargo labeled 'Closed Access', the embargo will last until 2025-05-01","The student, Karthik Venkat Ramanan, accepted the attached license on 2023-05-03 at 18:29.","The student, Karthik Venkat Ramanan, submitted this Thesis for approval on 2023-05-03 at 18:47.","This Thesis was approved for publication on 2023-05-04 at 11:07.","DSpace SAF Submission Ingestion Package generated from Vireo submission #19327 on 2023-09-01 at 17:22:23","Dynamic topic models (DTMs) play a crucial role in generating insights from large timestamped corpora of text by capturing the evolution of topics over time. Despite their popularity, existing DTMs are fully unsupervised, resulting in generated topic evolutions that often do not cater to a user’s needs. Additionally, the topic evolutions produced by DTMs tend to contain generic terms that do not accurately represent their designated time steps. This is particularly problematic as DTMs are frequently employed for analyzing the evolution of specific topics within a corpus. To address these challenges, we propose ReGenT, a framework for Dynamic, Discriminative Topic Discovery. This task aims to discover topic evolutions from temporal corpora that align with a set of user-provided category names while uniquely capturing topics at each time step. We accomplish this by (1) utilizing a retrieval-QA framework to retrieve relevant words for seeds with high granularity, (2) automatically generating and ranking strong questions to probe future words to expand our initial word set, (3) ensuring that the mined words are distinctly popular at a given time, and (4) iteratively refining our word list through ensemble ranking. We conduct experiments on two diverse datasets and demonstrate that ReGenT achieves state-of-the-art performance through extensive evaluations."]},{"key":"dc:format","label":"Dc Format","values":["application/pdf"]},{"key":"dc:title","label":"Title","values":["User-guided dynamic topic discovery in large texts"]}]}],"canonical_facts":{"dc:contributor":["Han, Jiawei"],"dc:creator":["Venkat Ramanan, Karthik"],"dc:date":["2023-05","2023-05-04"],"dc:description":["Submission published under a 24 month embargo labeled 'Closed Access', the embargo will last until 2025-05-01","The student, Karthik Venkat Ramanan, accepted the attached license on 2023-05-03 at 18:29.","The student, Karthik Venkat Ramanan, submitted this Thesis for approval on 2023-05-03 at 18:47.","This Thesis was approved for publication on 2023-05-04 at 11:07.","DSpace SAF Submission Ingestion Package generated from Vireo submission #19327 on 2023-09-01 at 17:22:23","Dynamic topic models (DTMs) play a crucial role in generating insights from large timestamped corpora of text by capturing the evolution of topics over time. Despite their popularity, existing DTMs are fully unsupervised, resulting in generated topic evolutions that often do not cater to a user’s needs. Additionally, the topic evolutions produced by DTMs tend to contain generic terms that do not accurately represent their designated time steps. This is particularly problematic as DTMs are frequently employed for analyzing the evolution of specific topics within a corpus. To address these challenges, we propose ReGenT, a framework for Dynamic, Discriminative Topic Discovery. This task aims to discover topic evolutions from temporal corpora that align with a set of user-provided category names while uniquely capturing topics at each time step. We accomplish this by (1) utilizing a retrieval-QA framework to retrieve relevant words for seeds with high granularity, (2) automatically generating and ranking strong questions to probe future words to expand our initial word set, (3) ensuring that the mined words are distinctly popular at a given time, and (4) iteratively refining our word list through ensemble ranking. We conduct experiments on two diverse datasets and demonstrate that ReGenT achieves state-of-the-art performance through extensive evaluations."],"dc:format":["application/pdf"],"dc:identifier":["https://hdl.handle.net/2142/120590"],"dc:language":["en","eng"],"dc:rights":["Copyright 2023 Karthik Venkat Ramanan"],"dc:subject":["Information Retrieval","Question Answering","Natural Language Processing","Data Mining","Pretrained Language Models","Topic Modeling"],"dc:title":["User-guided dynamic topic discovery in large texts"],"dc:type":["text"],"thesis:degree_discipline":["Computer Science"],"thesis:degree_level":["Thesis"],"thesis:degree_name":["M.S."],"thesis:institution_name":["University of Illinois at Urbana-Champaign"]},"updated_at":"2026-07-22T22:24:57Z"}