{"id":{"repo_id":"uiuc","oai_identifier":"oai:www.ideals.illinois.edu:2142/124428"},"canonical_url":"https://search.dev.ndltd.org/etd/uiuc/oai:www.ideals.illinois.edu:2142/124428","repository":{"repo_id":"uiuc","name":"University of Illinois - Urbana-Champaign","base_url":"https://www.ideals.illinois.edu/oai-pmh"},"display":{"title":"Efficient retrieval-augmented generation","abstract":"Submission original under an indefinite embargo labeled 'Open Access'. The submission was exported from vireo on 2024-09-16 without embargo terms","abstract_html":"Submission original under an indefinite embargo labeled &#x27;Open Access&#x27;. The submission was exported from vireo on 2024-09-16 without embargo terms","abstract_has_math":false,"creators":["Chen, Ziyi"],"institution":"University of Illinois at Urbana-Champaign","degree_name":"M.S.","degree_level":"Thesis","degree_discipline":"Computer Science","degree_department":null,"school":null,"contributors":["Chang, Kevin Chen-Chuan"],"advisors":[],"committee_chairs":[],"committee_members":[],"year":2024,"date_issued":"2024-05","date_published":"2024-05","updated_at":"2026-07-22T22:25:00Z","subjects":["Efficient Natural Language Processing","Retrieval-augmented Generation"],"languages":["en","eng"],"rights":["Copyright 2024 Ziyi Chen"],"rights_urls":[],"identifier_entries":[]},"links":{"outbound_url":"https://hdl.handle.net/2142/124428","outbound_label":"Handle","outbound_source":"dc:identifier"},"metadata_groups":[{"id":"people","label":"People","entries":[{"key":"dc:contributor","label":"Contributor","values":["Chang, Kevin Chen-Chuan"]},{"key":"dc:creator","label":"Author","values":["Chen, Ziyi"]}]},{"id":"academic_context","label":"Academic Context","entries":[{"key":"dc:date","label":"Dc Date","values":["2024-05","2024-05-01"]},{"key":"dc:type","label":"Dc Type","values":["text"]},{"key":"thesis:degree_discipline","label":"Discipline","values":["Computer Science"]},{"key":"thesis:degree_level","label":"Degree Level","values":["Thesis"]},{"key":"thesis:degree_name","label":"Degree Name","values":["M.S."]},{"key":"thesis:institution_name","label":"Thesis Institution Name","values":["University of Illinois at Urbana-Champaign"]}]},{"id":"subjects_keywords","label":"Subjects and Keywords","entries":[{"key":"dc:subject","label":"Dc Subject","values":["Efficient Natural Language Processing","Retrieval-augmented Generation"]}]},{"id":"language_rights","label":"Language and Rights","entries":[{"key":"dc:language","label":"Dc Language","values":["en","eng"]},{"key":"dc:rights","label":"Dc Rights","values":["Copyright 2024 Ziyi Chen"]}]},{"id":"identifiers","label":"Identifiers","entries":[{"key":"dc:identifier","label":"Identifier","values":["https://hdl.handle.net/2142/124428"]}]},{"id":"additional","label":"Additional Metadata","entries":[{"key":"dc:description","label":"Description","values":["Submission original under an indefinite embargo labeled 'Open Access'. The submission was exported from vireo on 2024-09-16 without embargo terms","The student, Ziyi Chen, accepted the attached license on 2024-04-28 at 15:42.","The student, Ziyi Chen, submitted this Thesis for approval on 2024-04-28 at 15:45.","This Thesis was approved for publication on 2024-05-01 at 10:12.","DSpace SAF Submission Ingestion Package generated from Vireo submission #20683 on 2024-09-16 at 00:37:20","Retrieval-Augmented Generation (RAG) is a technique to augment language models with external knowledge of corpus. Despite the rapid evolution of large language models, RAG is still a promising method for solving the difficulty of updating information and unreliable memorization of large language models as many research endeavors and commercial services leveraged retrieval-augmented generation to improve reliability. However, RAG has its drawbacks including high latency and intensive computational resource utilization. The inefficiency resides in two aspects: the long input due to retrieved documents and slow autoregressive generation. To address these two issues, we propose Efficient Title Reranker, a fast reranker to select important documents for input, and Cascade Speculative Drafting which improves upon speculative decoding to increase the generation efficiency of large language models. The Efficient Title Reranker achieves state-of-the-art in retrieval accuracy while being more efficient than the baseline on the KILT knowledge benchmark. On the other hand, Cascade Speculative Drafting outperforms Speculative Decoding in generation speed on both GSM8k and MMLU without additional training."]},{"key":"dc:format","label":"Dc Format","values":["application/pdf"]},{"key":"dc:title","label":"Title","values":["Efficient retrieval-augmented generation"]}]}],"canonical_facts":{"dc:contributor":["Chang, Kevin Chen-Chuan"],"dc:creator":["Chen, Ziyi"],"dc:date":["2024-05","2024-05-01"],"dc:description":["Submission original under an indefinite embargo labeled 'Open Access'. The submission was exported from vireo on 2024-09-16 without embargo terms","The student, Ziyi Chen, accepted the attached license on 2024-04-28 at 15:42.","The student, Ziyi Chen, submitted this Thesis for approval on 2024-04-28 at 15:45.","This Thesis was approved for publication on 2024-05-01 at 10:12.","DSpace SAF Submission Ingestion Package generated from Vireo submission #20683 on 2024-09-16 at 00:37:20","Retrieval-Augmented Generation (RAG) is a technique to augment language models with external knowledge of corpus. Despite the rapid evolution of large language models, RAG is still a promising method for solving the difficulty of updating information and unreliable memorization of large language models as many research endeavors and commercial services leveraged retrieval-augmented generation to improve reliability. However, RAG has its drawbacks including high latency and intensive computational resource utilization. The inefficiency resides in two aspects: the long input due to retrieved documents and slow autoregressive generation. To address these two issues, we propose Efficient Title Reranker, a fast reranker to select important documents for input, and Cascade Speculative Drafting which improves upon speculative decoding to increase the generation efficiency of large language models. The Efficient Title Reranker achieves state-of-the-art in retrieval accuracy while being more efficient than the baseline on the KILT knowledge benchmark. On the other hand, Cascade Speculative Drafting outperforms Speculative Decoding in generation speed on both GSM8k and MMLU without additional training."],"dc:format":["application/pdf"],"dc:identifier":["https://hdl.handle.net/2142/124428"],"dc:language":["en","eng"],"dc:rights":["Copyright 2024 Ziyi Chen"],"dc:subject":["Efficient Natural Language Processing","Retrieval-augmented Generation"],"dc:title":["Efficient retrieval-augmented generation"],"dc:type":["text"],"thesis:degree_discipline":["Computer Science"],"thesis:degree_level":["Thesis"],"thesis:degree_name":["M.S."],"thesis:institution_name":["University of Illinois at Urbana-Champaign"]},"updated_at":"2026-07-22T22:25:00Z"}