{"id":{"repo_id":"uoit","oai_identifier":"oai:ontariotechu.scholaris.ca:10155/1934"},"canonical_url":"https://search.dev.ndltd.org/etd/uoit/oai:ontariotechu.scholaris.ca:10155/1934","repository":{"repo_id":"uoit","name":"Ontario Institute of Technology","base_url":"https://ontariotechu.scholaris.ca/server/oai/request"},"display":{"title":"Beyond rules: how Large Language Models are redefining cryptographic misuse detection","abstract":"The use of Large Language Models (LLMs) in software development is rapidly growing, with developers increasingly relying on these models for coding assistance, including security-critical tasks. Our work presents a comprehensive comparison between traditional static analysis tools for cryptographic API misuse detection—CryptoGuard, CogniCrypt, and Snyk Code—and the LLMs—GPT, Llama, Claude, and Gemini. Using benchmark datasets (OWASP, CryptoAPI, and MASC), we evaluate the effectiveness of each tool in identifying cryptographic misuses. Our findings show that GPT 4-o-mini surpasses current state-of-the-art static analysis tools on the CryptoAPI and MASC datasets, though it lags on the OWASP dataset. Additionally, we assess the quality of LLM responses to determine which models provide actionable and accurate advice, giving developers insights into their practical utility for secure coding. This study highlights the comparative strengths and limitations of static analysis versus LLM-driven approaches, offering valuable insights into the evolving role of AI in advancing software security practices.","abstract_html":"The use of Large Language Models (LLMs) in software development is rapidly growing, with developers increasingly relying on these models for coding assistance, including security-critical tasks. Our work presents a comprehensive comparison between traditional static analysis tools for cryptographic API misuse detection—CryptoGuard, CogniCrypt, and Snyk Code—and the LLMs—GPT, Llama, Claude, and Gemini. Using benchmark datasets (OWASP, CryptoAPI, and MASC), we evaluate the effectiveness of each tool in identifying cryptographic misuses. Our findings show that GPT 4-o-mini surpasses current state-of-the-art static analysis tools on the CryptoAPI and MASC datasets, though it lags on the OWASP dataset. Additionally, we assess the quality of LLM responses to determine which models provide actionable and accurate advice, giving developers insights into their practical utility for secure coding. This study highlights the comparative strengths and limitations of static analysis versus LLM-driven approaches, offering valuable insights into the evolving role of AI in advancing software security practices.","abstract_has_math":false,"creators":["Masood, Zohaib"],"institution":"University of Ontario Institute of Technology","degree_name":"Master of Science (MSc)","degree_level":null,"degree_discipline":"Computer Science","degree_department":null,"school":null,"contributors":[],"advisors":["Vargas Martin, Miguel"],"committee_chairs":[],"committee_members":[],"year":2025,"date_issued":"2025-03-01","date_published":"2025-03-01","updated_at":"2026-07-24T05:35:28Z","subjects":[],"languages":["en"],"rights":[],"rights_urls":[],"identifier_entries":[]},"links":{"outbound_url":"https://hdl.handle.net/10155/1934","outbound_label":"Handle","outbound_source":"dc:identifier.uri"},"metadata_groups":[{"id":"people","label":"People","entries":[{"key":"dc:contributor.advisor","label":"Advisor","values":["Vargas Martin, Miguel"]},{"key":"dc:creator","label":"Author","values":["Masood, Zohaib"]}]},{"id":"academic_context","label":"Academic Context","entries":[{"key":"dc:date.accessioned","label":"Dc Date Accessioned","values":["2025-04-29T18:01:36Z"]},{"key":"dc:date.available","label":"Dc Date Available","values":["2025-04-29T18:01:36Z"]},{"key":"dc:date.issued","label":"Date","values":["2025-03-01"]},{"key":"dc:type","label":"Dc Type","values":["Thesis"]},{"key":"thesis:degree_discipline","label":"Discipline","values":["Computer Science"]},{"key":"thesis:degree_name","label":"Degree Name","values":["Master of Science (MSc)"]},{"key":"thesis:institution_name","label":"Thesis Institution Name","values":["University of Ontario Institute of Technology"]}]},{"id":"language_rights","label":"Language and Rights","entries":[{"key":"dc:language.iso","label":"Language (ISO)","values":["en"]}]},{"id":"identifiers","label":"Identifiers","entries":[{"key":"dc:identifier.uri","label":"Identifier URI","values":["https://hdl.handle.net/10155/1934"]}]},{"id":"additional","label":"Additional Metadata","entries":[{"key":"dc:description.abstract","label":"Abstract","values":["The use of Large Language Models (LLMs) in software development is rapidly growing, with developers increasingly relying on these models for coding assistance, including security-critical tasks. Our work presents a comprehensive comparison between traditional static analysis tools for cryptographic API misuse detection—CryptoGuard, CogniCrypt, and Snyk Code—and the LLMs—GPT, Llama, Claude, and Gemini. Using benchmark datasets (OWASP, CryptoAPI, and MASC), we evaluate the effectiveness of each tool in identifying cryptographic misuses. Our findings show that GPT 4-o-mini surpasses current state-of-the-art static analysis tools on the CryptoAPI and MASC datasets, though it lags on the OWASP dataset. Additionally, we assess the quality of LLM responses to determine which models provide actionable and accurate advice, giving developers insights into their practical utility for secure coding. This study highlights the comparative strengths and limitations of static analysis versus LLM-driven approaches, offering valuable insights into the evolving role of AI in advancing software security practices."]},{"key":"dc:title","label":"Title","values":["Beyond rules: how Large Language Models are redefining cryptographic misuse detection"]}]}],"canonical_facts":{"dc:contributor.advisor":["Vargas Martin, Miguel"],"dc:creator":["Masood, Zohaib"],"dc:date.accessioned":["2025-04-29T18:01:36Z"],"dc:date.available":["2025-04-29T18:01:36Z"],"dc:date.issued":["2025-03-01"],"dc:description.abstract":["The use of Large Language Models (LLMs) in software development is rapidly growing, with developers increasingly relying on these models for coding assistance, including security-critical tasks. Our work presents a comprehensive comparison between traditional static analysis tools for cryptographic API misuse detection—CryptoGuard, CogniCrypt, and Snyk Code—and the LLMs—GPT, Llama, Claude, and Gemini. Using benchmark datasets (OWASP, CryptoAPI, and MASC), we evaluate the effectiveness of each tool in identifying cryptographic misuses. Our findings show that GPT 4-o-mini surpasses current state-of-the-art static analysis tools on the CryptoAPI and MASC datasets, though it lags on the OWASP dataset. Additionally, we assess the quality of LLM responses to determine which models provide actionable and accurate advice, giving developers insights into their practical utility for secure coding. This study highlights the comparative strengths and limitations of static analysis versus LLM-driven approaches, offering valuable insights into the evolving role of AI in advancing software security practices."],"dc:identifier.uri":["https://hdl.handle.net/10155/1934"],"dc:language.iso":["en"],"dc:title":["Beyond rules: how Large Language Models are redefining cryptographic misuse detection"],"dc:type":["Thesis"],"thesis:degree_discipline":["Computer Science"],"thesis:degree_name":["Master of Science (MSc)"],"thesis:institution_name":["University of Ontario Institute of Technology"]},"updated_at":"2026-07-24T05:35:28Z"}