{"id":{"repo_id":"mit","oai_identifier":"oai:dspace.mit.edu:1721.1/142842"},"canonical_url":"https://search.dev.ndltd.org/etd/mit/oai:dspace.mit.edu:1721.1/142842","repository":{"repo_id":"mit","name":"MIT","base_url":"https://dspace.mit.edu/oai/request"},"display":{"title":"ChaperoNet: Distillation of Language Model Semantics to Folded Three-Dimensional Protein Structures","abstract":"Determining the structure of proteins has been a long-standing goal in biology. Lan- guage models have been recently deployed to capture the evolutionary semantics of protein sequences, and as an emergent property, were found to be structural learn- ers. Enriched with multiple sequence alignments (MSA), these transformer models were able to capture significant information about a protein’s tertiary structure. In this work, we show how such structural information can be recovered by processing language model embeddings, and introduce a two-stage folding pipeline to directly es- timate three-dimensional folded structures from protein sequences. We envision that this pipeline will provide a basis for efficient, end-to-end protein structure prediction through protein language modeling.","abstract_html":"Determining the structure of proteins has been a long-standing goal in biology. Lan- guage models have been recently deployed to capture the evolutionary semantics of protein sequences, and as an emergent property, were found to be structural learn- ers. Enriched with multiple sequence alignments (MSA), these transformer models were able to capture significant information about a protein’s tertiary structure. In this work, we show how such structural information can be recovered by processing language model embeddings, and introduce a two-stage folding pipeline to directly es- timate three-dimensional folded structures from protein sequences. We envision that this pipeline will provide a basis for efficient, end-to-end protein structure prediction through protein language modeling.","abstract_has_math":false,"creators":["dos Santos Costa, Allan"],"institution":"Massachusetts Institute of Technology","degree_name":"Master","degree_level":null,"degree_discipline":null,"degree_department":"Program in Media Arts and Sciences (Massachusetts Institute of Technology)","school":null,"contributors":[],"advisors":["Jacobson, Joseph M."],"committee_chairs":[],"committee_members":[],"year":2021,"date_issued":"2021-09","date_published":"2021-09","updated_at":"2026-07-22T22:21:35Z","subjects":[],"languages":[],"rights":["In Copyright - Educational Use Permitted","Copyright MIT"],"rights_urls":["http://rightsstatements.org/page/InC-EDU/1.0/"],"identifier_entries":[]},"links":{"outbound_url":"https://hdl.handle.net/1721.1/142842","outbound_label":"Handle","outbound_source":"dc:identifier.uri"},"metadata_groups":[{"id":"people","label":"People","entries":[{"key":"dc:contributor.advisor","label":"Advisor","values":["Jacobson, Joseph M."]},{"key":"dc:contributor.department","label":"Department","values":["Program in Media Arts and Sciences (Massachusetts Institute of Technology)"]},{"key":"dc:creator","label":"Author","values":["dos Santos Costa, Allan"]}]},{"id":"academic_context","label":"Academic Context","entries":[{"key":"dc:date.accessioned","label":"Dc Date Accessioned","values":["2022-05-31T13:32:10Z"]},{"key":"dc:date.available","label":"Dc Date Available","values":["2022-05-31T13:32:10Z"]},{"key":"dc:date.issued","label":"Date","values":["2021-09"]},{"key":"dc:publisher","label":"Institution","values":["Massachusetts Institute of Technology"]},{"key":"dc:type","label":"Dc Type","values":["Thesis"]},{"key":"thesis:degree_name","label":"Degree Name","values":["Master","Master of Science in Media Arts and Sciences"]}]},{"id":"language_rights","label":"Language and Rights","entries":[{"key":"dc:rights","label":"Dc Rights","values":["In Copyright - Educational Use Permitted","Copyright MIT"]},{"key":"dc:rights.uri","label":"Rights URI","values":["http://rightsstatements.org/page/InC-EDU/1.0/"]}]},{"id":"identifiers","label":"Identifiers","entries":[{"key":"dc:identifier.uri","label":"Identifier URI","values":["https://hdl.handle.net/1721.1/142842"]}]},{"id":"additional","label":"Additional Metadata","entries":[{"key":"dc:description.abstract","label":"Abstract","values":["Determining the structure of proteins has been a long-standing goal in biology. Lan- guage models have been recently deployed to capture the evolutionary semantics of protein sequences, and as an emergent property, were found to be structural learn- ers. Enriched with multiple sequence alignments (MSA), these transformer models were able to capture significant information about a protein’s tertiary structure. In this work, we show how such structural information can be recovered by processing language model embeddings, and introduce a two-stage folding pipeline to directly es- timate three-dimensional folded structures from protein sequences. We envision that this pipeline will provide a basis for efficient, end-to-end protein structure prediction through protein language modeling."]},{"key":"dc:description.degree","label":"Dc Description Degree","values":["S.M."]},{"key":"dc:title","label":"Title","values":["ChaperoNet: Distillation of Language Model Semantics to Folded Three-Dimensional Protein Structures"]}]}],"canonical_facts":{"dc:contributor.advisor":["Jacobson, Joseph M."],"dc:contributor.department":["Program in Media Arts and Sciences (Massachusetts Institute of Technology)"],"dc:creator":["dos Santos Costa, Allan"],"dc:date.accessioned":["2022-05-31T13:32:10Z"],"dc:date.available":["2022-05-31T13:32:10Z"],"dc:date.issued":["2021-09"],"dc:description.abstract":["Determining the structure of proteins has been a long-standing goal in biology. Lan- guage models have been recently deployed to capture the evolutionary semantics of protein sequences, and as an emergent property, were found to be structural learn- ers. Enriched with multiple sequence alignments (MSA), these transformer models were able to capture significant information about a protein’s tertiary structure. In this work, we show how such structural information can be recovered by processing language model embeddings, and introduce a two-stage folding pipeline to directly es- timate three-dimensional folded structures from protein sequences. We envision that this pipeline will provide a basis for efficient, end-to-end protein structure prediction through protein language modeling."],"dc:description.degree":["S.M."],"dc:identifier.uri":["https://hdl.handle.net/1721.1/142842"],"dc:publisher":["Massachusetts Institute of Technology"],"dc:rights":["In Copyright - Educational Use Permitted","Copyright MIT"],"dc:rights.uri":["http://rightsstatements.org/page/InC-EDU/1.0/"],"dc:title":["ChaperoNet: Distillation of Language Model Semantics to Folded Three-Dimensional Protein Structures"],"dc:type":["Thesis"],"thesis:degree_name":["Master","Master of Science in Media Arts and Sciences"]},"updated_at":"2026-07-22T22:21:35Z"}