<?xml version="1.0" encoding="UTF-8"?><article xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="pmc-domain-id">102</journal-id><journal-id journal-id-type="pmc-domain">genores</journal-id><journal-title-group><journal-title>Genome Research</journal-title><abbrev-journal-title>Genome Res</abbrev-journal-title></journal-title-group><publisher><publisher-name>Cold Spring Harbor Laboratory Press</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="pmcid">PMC11368176</article-id><article-id pub-id-type="pmcaid">11368176</article-id><article-id pub-id-type="pmcaiid">11368176</article-id><article-id pub-id-type="pmid">38951026</article-id><article-id pub-id-type="doi">10.1101/gr.278870.123</article-id><title-group><article-title>CodonBERT large language model for mRNA vaccines</article-title></title-group><contrib-group content-type="author"><contrib><name name-style="western"><surname>Li</surname><given-names initials="S">Sizhen</given-names></name><xref ref-type="aff" rid="af1">1</xref><xref rid="FN1" ref-type="author-notes">4</xref></contrib><contrib><name name-style="western"><surname>Moayedpour</surname><given-names initials="S">Saeed</given-names></name><xref ref-type="aff" rid="af1">1</xref><xref rid="FN1" ref-type="author-notes">4</xref></contrib><contrib><name name-style="western"><surname>Li</surname><given-names initials="R">Ruijiang</given-names></name><xref ref-type="aff" rid="af1">1</xref></contrib><contrib><name name-style="western"><surname>Bailey</surname><given-names initials="M">Michael</given-names></name><xref ref-type="aff" rid="af1">1</xref></contrib><contrib><name name-style="western"><surname>Riahi</surname><given-names initials="S">Saleh</given-names></name><xref ref-type="aff" rid="af1">1</xref></contrib><contrib><name name-style="western"><surname>Kogler-Anele</surname><given-names initials="L">Lorenzo</given-names></name><xref ref-type="aff" rid="af1">1</xref></contrib><contrib><name name-style="western"><surname>Miladi</surname><given-names initials="M">Milad</given-names></name><xref ref-type="aff" rid="af2">2</xref></contrib><contrib><name name-style="western"><surname>Miner</surname><given-names initials="J">Jacob</given-names></name><xref ref-type="aff" rid="af2">2</xref></contrib><contrib><name name-style="western"><surname>Pertuy</surname><given-names initials="F">Fabien</given-names></name><xref ref-type="aff" rid="af2">2</xref></contrib><contrib><name name-style="western"><surname>Zheng</surname><given-names initials="D">Dinghai</given-names></name><xref ref-type="aff" rid="af2">2</xref></contrib><contrib><name name-style="western"><surname>Wang</surname><given-names initials="J">Jun</given-names></name><xref ref-type="aff" rid="af2">2</xref></contrib><contrib><name name-style="western"><surname>Balsubramani</surname><given-names initials="A">Akshay</given-names></name><xref ref-type="aff" rid="af2">2</xref></contrib><contrib><name name-style="western"><surname>Tran</surname><given-names initials="K">Khang</given-names></name><xref ref-type="aff" rid="af2">2</xref></contrib><contrib><name name-style="western"><surname>Zacharia</surname><given-names initials="M">Minnie</given-names></name><xref ref-type="aff" rid="af2">2</xref></contrib><contrib><name name-style="western"><surname>Wu</surname><given-names initials="M">Monica</given-names></name><xref ref-type="aff" rid="af2">2</xref></contrib><contrib><name name-style="western"><surname>Gu</surname><given-names initials="X">Xiaobo</given-names></name><xref ref-type="aff" rid="af2">2</xref></contrib><contrib><name name-style="western"><surname>Clinton</surname><given-names initials="R">Ryan</given-names></name><xref ref-type="aff" rid="af2">2</xref></contrib><contrib><name name-style="western"><surname>Asquith</surname><given-names initials="C">Carla</given-names></name><xref ref-type="aff" rid="af2">2</xref></contrib><contrib><name name-style="western"><surname>Skaleski</surname><given-names initials="J">Joseph</given-names></name><xref ref-type="aff" rid="af2">2</xref></contrib><contrib><name name-style="western"><surname>Boeglin</surname><given-names initials="L">Lianne</given-names></name><xref ref-type="aff" rid="af2">2</xref></contrib><contrib><name name-style="western"><surname>Chivukula</surname><given-names initials="S">Sudha</given-names></name><xref ref-type="aff" rid="af2">2</xref></contrib><contrib><name name-style="western"><surname>Dias</surname><given-names initials="A">Anusha</given-names></name><xref ref-type="aff" rid="af2">2</xref></contrib><contrib><name name-style="western"><surname>Strugnell</surname><given-names initials="T">Tod</given-names></name><xref ref-type="aff" rid="af2">2</xref></contrib><contrib><name name-style="western"><surname>Montoya</surname><given-names initials="FU">Fernando Ulloa</given-names></name><xref ref-type="aff" rid="af3">3</xref></contrib><contrib><name name-style="western"><surname>Agarwal</surname><given-names initials="V">Vikram</given-names></name><xref ref-type="aff" rid="af2">2</xref></contrib><contrib><name name-style="western"><surname>Bar-Joseph</surname><given-names initials="Z">Ziv</given-names></name><xref ref-type="aff" rid="af1">1</xref><xref ref-type="author-notes" rid="_fncrsp93pmc__">✉</xref></contrib><contrib><name name-style="western"><surname>Jager</surname><given-names initials="S">Sven</given-names></name><xref ref-type="aff" rid="af1">1</xref></contrib></contrib-group><aff id="af1"><label>1</label>Digital R&amp;D, Sanofi, Cambridge, Massachusetts 02141, USA;</aff><aff id="af2"><label>2</label>mRNA Center of Excellence, Sanofi, Waltham, Massachusetts 02451, USA;</aff><aff id="af3"><label>3</label>mRNA Center of Excellence, Sanofi, 69280 Marcy L'Etoile, France</aff><author-notes><fn id="FN1"><label>4</label><p>These authors contributed equally to this work.</p></fn><fn id="corresp1"><label>✉</label><p>Corresponding authors: <email>zivbj@cs.cmu.edu</email>, <email>sven.jager@sanofi.com</email></p></fn><fn id="_fncrsp93pmc__"><label>✉</label><p>Corresponding author.</p></fn></author-notes><pub-date><month>7</month><year>2024</year></pub-date><volume>34</volume><issue>7</issue><fpage>1027</fpage><page-range>1027–1035</page-range><pub-history><event event-type="pmc-release"><date><day>1</day><month>1</month><year>2025</year></date></event></pub-history><permissions><copyright-statement>
<ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="http://genome.cshlp.org/site/misc/terms.xhtml" ext-link-type="uri">© 2024 Li et al.; Published by Cold Spring Harbor Laboratory Press</ext-link></copyright-statement><license><license-p>This article is distributed exclusively by Cold Spring Harbor Laboratory Press for the first six months after the full-issue publication date (see <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://genome.cshlp.org/site/misc/terms.xhtml" ext-link-type="uri">https://genome.cshlp.org/site/misc/terms.xhtml</ext-link>). After six months, it is available under a Creative Commons License (Attribution-NonCommercial 4.0 International), as described at <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://creativecommons.org/licenses/by-nc/4.0/" ext-link-type="uri">http://creativecommons.org/licenses/by-nc/4.0/</ext-link>.</license-p></license></permissions><self-uri xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="1027.pdf" content-type="pmc-pdf"><?cloudpmc-path d27d/11368176/27b81d639f18/1027.pdf?><?cloudpmc-bucket app?><?size 3210006?></self-uri><abstract id="abstract1"><title>Abstract</title><p>mRNA-based vaccines and therapeutics are gaining popularity and usage across a wide range of conditions. One of the critical issues when designing such mRNAs is sequence optimization. Even small proteins or peptides can be encoded by an enormously large number of mRNAs. The actual mRNA sequence can have a large impact on several properties, including expression, stability, immunogenicity, and more. To enable the selection of an optimal sequence, we developed CodonBERT, a large language model (LLM) for mRNAs. Unlike prior models, CodonBERT uses codons as inputs, which enables it to learn better representations. CodonBERT was trained using more than 10 million mRNA sequences from a diverse set of organisms. The resulting model captures important biological concepts. CodonBERT can also be extended to perform prediction tasks for various mRNA properties. CodonBERT outperforms previous mRNA prediction methods, including on a new flu vaccine data set.</p></abstract><custom-meta-group><custom-meta><meta-name>status</meta-name><meta-value>released</meta-value></custom-meta><custom-meta><meta-name>display-pdf</meta-name><meta-value>yes</meta-value></custom-meta><custom-meta><meta-name>is-olf</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>is-manuscript</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>is-preprint</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>is-journal-matter</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>is-scanned</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>is-retracted</meta-name><meta-value>no</meta-value></custom-meta></custom-meta-group></article-meta><notes notes-type="article-notes"><sec id="historyarticle-meta1" sec-type="history" disp-level="2"><p>Received 2023 Dec 15; Accepted 2024 Jun 25.</p></sec></notes></front><body><p>mRNA vaccines have emerged as a high-potency, fast-production, low-cost, and safe alternative to traditional vaccines (<xref rid="GR278870LIC32" ref-type="bibr">Pardi et al. 2018</xref>, <xref rid="GR278870LIC33" ref-type="bibr">2020</xref>; <xref rid="GR278870LIC49" ref-type="bibr">Zhang et al. 2019</xref>; <xref rid="GR278870LIC17" ref-type="bibr">Jackson et al. 2020b</xref>). mRNA vaccines are currently being developed for a broad range of human viruses and bacteria, including SARS-CoV-2, influenza, Zika, chlamydia, and more (<xref rid="GR278870LIC23" ref-type="bibr">Maruggi et al. 2017</xref>; <xref rid="GR278870LIC31" ref-type="bibr">Pardi et al. 2017</xref>; <xref rid="GR278870LIC16" ref-type="bibr">Jackson et al. 2020a</xref>; <xref rid="GR278870LIC36" ref-type="bibr">Pilkington et al. 2021</xref>). They are also being investigated as potential treatments for several diseases, including lung cancer, breast cancer, and melanoma (<xref rid="GR278870LIC28" ref-type="bibr">Miao et al. 2021</xref>; <xref rid="GR278870LIC22" ref-type="bibr">Lorentzen et al. 2022</xref>).</p><p>The expression level of a vaccine directly affects its potency, ultimate immunogenicity, and efficacy (<xref rid="GR278870LIC41" ref-type="bibr">Schlake et al. 2012</xref>). The higher the level of expression of the antigenic protein encoded by the mRNA sequence, the smaller amount of the vaccine is needed to achieve the desired immune response, which can make the vaccine more cost-effective and easier to manufacture (<xref rid="GR278870LIC32" ref-type="bibr">Pardi et al. 2018</xref>). Consequently, using a lower dose can help reduce reactogenicity (<xref rid="GR278870LIC3" ref-type="bibr">Ahmad et al. 2022</xref>) and maintain the immune response over a longer period (<xref rid="GR278870LIC20" ref-type="bibr">Leppek et al. 2022</xref>), leading to better safety and efficacy.</p><p>A human protein with an average length of 500 amino acids can be encoded by roughly 3<sup>500</sup> different codon sequences. Although only one of those is encoded in the virus or DNA of interest, this is not necessarily the optimal sequence for a vaccine. The classical method to find the optimal mRNA sequence is codon optimization, which selects the most optimal codon for each amino acid using the codon bias in the host organism (<xref rid="GR278870LIC26" ref-type="bibr">Mauro and Chappell 2014</xref>). This method has been widely applied, including for optimizing recombinant protein drugs, nucleic acid therapies, gene therapy, mRNA therapy, and DNA/RNA vaccines (<xref rid="GR278870LIC5" ref-type="bibr">Al-Hawash et al. 2017</xref>; <xref rid="GR278870LIC46" ref-type="bibr">Webster et al. 2017</xref>; <xref rid="GR278870LIC25" ref-type="bibr">Mauro 2018</xref>). However, codon optimization alone does not consider several key properties that impact protein expression (<xref rid="GR278870LIC34" ref-type="bibr">Parret et al. 2016</xref>). For instance, RNA structural properties (e.g., stem loops and pseudoknots) have been shown to play a major role for noncoding RNAs (such as riboswitches or aptamers) (<xref rid="GR278870LIC13" ref-type="bibr">Groher et al. 2018</xref>; <xref rid="GR278870LIC42" ref-type="bibr">Schmidt et al. 2020</xref>).</p><p>Although mRNA sequence heavily influences cellular RNA stability (<xref rid="GR278870LIC2" ref-type="bibr">Agarwal and Shendure 2020</xref>; <xref rid="GR278870LIC1" ref-type="bibr">Agarwal and Kelley 2022</xref>), secondary structure can also impact mRNA stability in solution and modulate protein expression (<xref rid="GR278870LIC24" ref-type="bibr">Mauger et al. 2019</xref>; <xref rid="GR278870LIC20" ref-type="bibr">Leppek et al. 2022</xref>; <xref rid="GR278870LIC30" ref-type="bibr">Nieuwkoop et al. 2023</xref>; <xref rid="GR278870LIC50" ref-type="bibr">Zhang et al. 2023</xref>). For example, replacing a codon with a synonymous codon can alter the local base-pairing interactions and affect nearby structural motifs (<xref rid="GR278870LIC14" ref-type="bibr">Groher et al. 2019</xref>; <xref rid="GR278870LIC21" ref-type="bibr">Li et al. 2021</xref>). Thus, optimizing each codon independently is not sufficient to generate highly expressed proteins.</p><p>Pretraining a large language model (LLM) based on large-scale unlabeled text, followed by fine-tuning, has been widely adopted for natural language processing (<xref rid="GR278870LIC35" ref-type="bibr">Peters et al. 2018</xref>; <xref rid="GR278870LIC37" ref-type="bibr">Radford et al. 2018</xref>; <xref rid="GR278870LIC9" ref-type="bibr">Devlin et al. 2019</xref>). Recently, this concept has been scaled to biological sequences (protein, DNA, and RNA) (<xref rid="GR278870LIC7" ref-type="bibr">Bepler and Berger 2021</xref>; <xref rid="GR278870LIC18" ref-type="bibr">Ji et al. 2021</xref>; <xref rid="GR278870LIC40" ref-type="bibr">Rives et al. 2021</xref>; <xref rid="GR278870LIC4" ref-type="bibr">Akiyama and Sakakibara 2022</xref>; <xref rid="GR278870LIC8" ref-type="bibr">Chen et al. 2022</xref>). Such models can be used to embed nucleotides and use these embeddings for downstream supervised learning tasks. However, as we show, such LLMs may not be ideal for predicting protein expression owing to their focus on individual nucleotides and noncoding regions. More recent work, such as cdsBERT (<xref rid="GR278870LIC15" ref-type="bibr">Hallee et al. 2023</xref>), addresses the issue of codon awareness for a protein language model.</p><p>To address these limitations, we developed CodonBERT, an LLM that extends the BERT model and applies it to the language of mRNAs. CodonBERT uses a multihead attention transformer architecture framework. The pretrained model can also be generalized to a diverse set of supervised learning tasks. We pretrained CodonBERT using 10 million mRNA coding sequences (CDSs) spanning an evolutionarily diverse set of organisms. Next, we used it to perform several mRNA prediction tasks, including protein expression and mRNA degradation prediction. As we show, both the pretrained and the fine-tuned version of the models can learn new biology and improve on current state-of-the-art methods for mRNA vaccine design.</p><p>To assess generalization of our CodonBERT model, we collected a novel hemagglutinin flu vaccine data set. Different mRNA candidates that encode the influenza hemagglutinin antigen (i.e., with fixed untranslated regions and a variable coding region) were designed, synthesized, and transfected into cells. The protein expression levels corresponding to these mRNA sequences were measured and used as labels for a supervised learning task. CodonBERT leads to better performance than existing methods.</p><sec id="s1" disp-level="1"><title>Results</title><p>We developed a LLM, CodonBERT, for mRNA analysis and prediction tasks. CodonBERT was pretrained using 10 million mRNA sequences derived from mammals, bacteria, and human viruses. All sequences were hierarchically labeled using 14 categories as shown in <xref rid="GR278870LIF1" ref-type="fig">Figure 1</xref>A. CodonBERT takes the coding region as input, using codons as tokens, and outputs an embedding that provides contextual codon representations. The embeddings provided by CodonBERT can be combined with additional trainable layers to perform various downstream regression and prediction tasks, including the prediction of protein expression and mRNA degradation.</p><fig id="GR278870LIF1" position="float"><?disp-level 2?><label>Figure 1.</label><caption><p>Pretraining data distribution and CodonBERT model architecture. (<italic>A</italic>) Hierarchically classified mRNA sequences for pretraining. All the 14 leaf-level classes (those annotated with an asterisk are numbered). The angle of each segment is proportional to the number of sequences belonging to this group. (<italic>B</italic>) Model architecture and training scheme deployed for two tasks of CodonBERT. (<italic>C</italic>) A stack of 12 transformer blocks employed in CodonBERT model.</p></caption><alternatives><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="image" xlink:href="1027f01.jpg"><?cloudpmc-path blobs/d27d/11368176/119abd0b2fcc/1027f01.jpg?><?cloudpmc-bucket cdn?><?image-server-status LOAD_COMPLETED?><?original-height 841?><?original-width 2000?><?scaled-height 336?><?scaled-width 800?></graphic><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="thumb" xlink:href="1027f01.gif"><?cloudpmc-path blobs/d27d/11368176/8ae0bdf39085/1027f01.gif?><?cloudpmc-bucket cdn?></graphic></alternatives></fig><p>A schematic representation of CodonBERT's architecture is provided in <xref rid="GR278870LIF1" ref-type="fig">Figure 1</xref>B. We pretrained CodonBERT with two tasks: masked language model (MLM) learning and sequence taxonomy prediction (STP). The MLM task learns the codon representation, interactions between codons, and relationships between codons and sequences. The STP task aims to directly model the sequence representation and understand the evolutionary relationships between mRNA sequences. In short, a pair of mRNA sequences, which is randomly sampled from either the same or different categories, is codon-tokenized, concatenated, and randomly masked. The masked inputs are further encoded with codon-, position-, and segment-based embeddings and fed into a stack of transformer layers using a multihead attention network with residual connections. CodonBERT is self-supervised and relies on masked token prediction and taxonomic sequence prediction for optimizing parameters (Methods).</p><sec id="s1a" disp-level="2"><title>Pretrained representation model</title><p>To assess our pretrained CodonBERT model, we built a held-out data set by randomly leaving out 1% of mRNA sequences for each category and trained the model with the remaining sequences. As illustrated in <xref rid="SD1" ref-type="supplementary-material">Supplemental Figure S1</xref>, during the pretraining phase, the model performance on two tasks (MLM and STP) substantially improves on both the training and evaluation sets. For example, the entropy loss of the MLM task (ℒ<sub>MLM</sub>) decreased to 2.85 on the evaluation set, which means that the model was able to narrow down the choice from 64 (uniform distribution) to only eight codons for each masked position (Methods).</p></sec><sec id="s1b" disp-level="2"><title>CodonBERT learns, on its own, the genetic code and evolutionary homology</title><p>In addition to the quantitative evaluation of model predictions, for example, loss and accuracy, we also performed several qualitative analyses of the embeddings provided by CodonBERT. To decipher what kind of biological information has been learned by the model and encoded in the representation, we randomly sampled 500 sequences for each category from the held-out data set and extracted high-dimensional codon and sequence embeddings from CodonBERT. These were projected onto a two-dimensional space (2D) by UMAP (<xref rid="GR278870LIC27" ref-type="bibr">McInnes et al. 2018</xref>).</p><p>In <xref rid="GR278870LIF2" ref-type="fig">Figure 2</xref>, A and B, each dot represents a codon and is annotated with different colors based on its type of codon and amino acid. Codons that encode the same amino acid, namely, synonymous codons, are spatially close to each other in <xref rid="GR278870LIF2" ref-type="fig">Figure 2</xref>B, which implies that CodonBERT learns the genetic code from the large-scale training set. For example, the amino acid valine, whose one-letter code is V, can be encoded by four codons: {GUA, GUU, GUG, GUC}. <xref rid="GR278870LIF2" ref-type="fig">Figure 2</xref>B illustrates four separate gray clusters for four possible codons, and four clusters are close to each other. We applied the <italic>k</italic>-nearest neighbor algorithm (kNN) on the output codon embeddings directly. Five hundred embeddings were sampled for each codon; 99.4% of the 500-nearest neighbors are the same codons. For the remaining misclassified codons, 87.7% are stop codons and are classified to other stop codons.</p><fig id="GR278870LIF2" position="float"><?disp-level 3?><label>Figure 2.</label><caption><p>Genetic code and evolutionary taxonomy information learned by the pretrained, unsupervised CodonBERT model. High-dimensional embeddings were projected into two-dimensional space using UMAP (<xref rid="GR278870LIC27" ref-type="bibr">McInnes et al. 2018</xref>). (<italic>A</italic>,<italic>B</italic>) Projected codon embeddings from the pretrained CodonBERT model. Each point represents a codon with different contexts, and its color corresponds to the type of codon (<italic>A</italic>) or amino acid (<italic>B</italic>) accordingly. (<italic>C</italic>) Projected sequence embedding from the pretrained CodonBERT model. Each point is a mRNA sequence, and its color represents the sequence label. (<italic>D</italic>) Projected codon embedding from the pretrained Codon2vec model. Each point shows a codon, and its color is the corresponding amino acid.</p></caption><alternatives><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="image" xlink:href="1027f02.jpg"><?cloudpmc-path blobs/d27d/11368176/2b6931facbce/1027f02.jpg?><?cloudpmc-bucket cdn?><?image-server-status LOAD_COMPLETED?><?original-height 1693?><?original-width 1898?><?scaled-height 677?><?scaled-width 759?></graphic><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="thumb" xlink:href="1027f02.gif"><?cloudpmc-path blobs/d27d/11368176/326189fb85ab/1027f02.gif?><?cloudpmc-bucket cdn?></graphic></alternatives></fig><p>Codon2vec, a Word2vec (<xref rid="GR278870LIC29" ref-type="bibr">Mikolov et al. 2013</xref>) model trained on the collected mRNA sequences, can also produce codon representations (Methods). However, compared with the codon representation generated by CodonBERT, the embedding of each codon from Codon2vec is fixed regardless of the context surrounding the codon (<xref rid="GR278870LIF2" ref-type="fig">Fig. 2</xref>D). This results in clusters that are often less accurate than the projection of codon representation from CodonBERT.</p><p>In addition to codon representation, CodonBERT also optimizes for sequence identification. 2D projections of the sequence embeddings of the held-out data set are presented. <xref rid="GR278870LIF2" ref-type="fig">Figure 2</xref>C illustrates clusters of four high-level sequence categories: (<italic>Escherichia coli</italic>, human virus, yeast, and mammal). Sequences from the same organism are clustered together with clear boundaries between the taxonomy classes. However, CodonBERT does not reveal a clear separation between different families within mammals as illustrated in <xref rid="SD1" ref-type="supplementary-material">Supplemental Figure S2A</xref>. This observation could be attributed to the similar codon usage patterns within the taxonomic groups. Codon usage, a critical factor in the translation efficiency of genes, can significantly influence the clustering of genetic sequences in computational analyses. We conducted a statistical analysis on the frequency of codons in different taxonomic groups and computed the Kullback–Leibler divergence of the codon usage between any two organisms (<xref rid="SD1" ref-type="supplementary-material">Supplemental Fig. S2B</xref>). We found that all taxonomic groups in mammals exhibit significant different codon usage compared with human virus, <italic>E. coli,</italic> and yeast. However, among mammals, the codon usage is too similar to distinguish.</p></sec><sec id="s1c" disp-level="2"><title>Evaluating CodonBERT and comparison to prior methods on supervised learning tasks</title><p>CodonBERT can be extended to perform supervised learning for specific mRNA prediction tasks. To evaluate the use of our LLM for downstream tasks and to compare it to prior methods, we collected several mRNA prediction data sets. <xref rid="GR278870LITB1" ref-type="table">Table 1</xref> presents the data sets and the mRNA properties. As can be seen, these included a diverse set of downstream tasks related to mRNA translation, stability, and regulation. In addition, these data sets represent a range of molecules, including newly published data sets for recombinant protein, bio-computing, and SARS-CoV-2 vaccine design. Finally, we generated a new data set to test CodonBERT in the context of mRNAs encoding the influenza hemagglutinin antigen for flu vaccines.</p><table-wrap id="GR278870LITB1" position="float"><?disp-level 3?><label>Table 1.</label><caption><p>The collection of the data sets with their corresponding mRNA source and property used for method evaluation</p></caption><table frame="hsides" rules="groups"><colgroup span="1"><col align="left" span="1"/><col align="left" span="1"/><col align="left" span="1"/><col align="char" char="." span="1"/><col align="center" span="1"/></colgroup><thead><tr><th align="left" rowspan="1" colspan="1">Data set</th><th align="center" rowspan="1" colspan="1">Target</th><th align="center" rowspan="1" colspan="1">Category</th><th align="center" rowspan="1" colspan="1">No. of mRNAs</th><th align="center" rowspan="1" colspan="1">Seq length</th></tr></thead><tbody><tr><td rowspan="1" colspan="1">MLOS flu vaccines (Sanofi-Aventis)</td><td rowspan="1" colspan="1">Expression</td><td rowspan="1" colspan="1">Regression</td><td rowspan="1" colspan="1">543</td><td rowspan="1" colspan="1">1698–1704</td></tr><tr><td rowspan="1" colspan="1">mRFP expression (<xref rid="GR278870LIC30" ref-type="bibr">Nieuwkoop et al. 2023</xref>)</td><td rowspan="1" colspan="1">Expression</td><td rowspan="1" colspan="1">Regression</td><td rowspan="1" colspan="1">1459</td><td rowspan="1" colspan="1">678–678</td></tr><tr><td rowspan="1" colspan="1">Fungal expression (<xref rid="GR278870LIC48" ref-type="bibr">Wint et al. 2022</xref>)</td><td rowspan="1" colspan="1">Expression</td><td rowspan="1" colspan="1">Regression</td><td rowspan="1" colspan="1">7056</td><td rowspan="1" colspan="1">150–3000</td></tr><tr><td rowspan="1" colspan="1"><italic>E. coli</italic> proteins (<xref rid="GR278870LIC11" ref-type="bibr">Ding et al. 2022</xref>)</td><td rowspan="1" colspan="1">Expression</td><td rowspan="1" colspan="1">Classification</td><td rowspan="1" colspan="1">6348</td><td rowspan="1" colspan="1">171–3000</td></tr><tr><td rowspan="1" colspan="1">Tc-riboswitches (<xref rid="GR278870LIC14" ref-type="bibr">Groher et al. 2019</xref>)</td><td rowspan="1" colspan="1">Switching factor</td><td rowspan="1" colspan="1">Regression</td><td rowspan="1" colspan="1">355</td><td rowspan="1" colspan="1">67–73</td></tr><tr><td rowspan="1" colspan="1">mRNA stability (<xref rid="GR278870LIC10" ref-type="bibr">Diez et al. 2022</xref>)</td><td rowspan="1" colspan="1">Stability</td><td rowspan="1" colspan="1">Regression</td><td rowspan="1" colspan="1">41,123</td><td rowspan="1" colspan="1">30–1497</td></tr><tr><td rowspan="1" colspan="1">SARS-CoV-2 vaccine degradation (<xref rid="GR278870LIC45" ref-type="bibr">Wayment-Steele et al. 2022</xref>)</td><td rowspan="1" colspan="1">Degradation</td><td rowspan="1" colspan="1">Regression</td><td rowspan="1" colspan="1">2400</td><td rowspan="1" colspan="1">81–81</td></tr></tbody></table><table-wrap-foot><fn id="fn2"><p>Each data set is split into training, validation, and test with a 0.7, 0.15, and 0.15 ratio. All the methods were optimized on the same data split.</p></fn></table-wrap-foot></table-wrap><p>The <bold>mRFP expression</bold> data set (<xref rid="GR278870LIC30" ref-type="bibr">Nieuwkoop et al. 2023</xref>) profiles protein production levels for several gene variants in <italic>E. coli</italic>. The <bold>fungal expression</bold> data set (<xref rid="GR278870LIC12" ref-type="bibr">Grigoriev et al. 2014</xref>; <xref rid="GR278870LIC48" ref-type="bibr">Wint et al. 2022</xref>) includes CDSs &gt;150 bp from a wide range of fungal genomes. The <bold><italic>E. coli</italic> protein</bold> data set (<xref rid="GR278870LIC11" ref-type="bibr">Ding et al. 2022</xref>) comprises experimental data for protein expression in <italic>E. coli</italic>, which are labeled as low, medium, or high expression (2308, 2067, and 1973 mRNA sequences, respectively). The <bold>mRNA stability</bold> data set (<xref rid="GR278870LIC10" ref-type="bibr">Diez et al. 2022</xref>) includes thousands of mRNA stability profiles obtained from human, mouse, frog, and fish. The <bold>Tc-riboswitch</bold> data set (<xref rid="GR278870LIC14" ref-type="bibr">Groher et al. 2019</xref>) consists of a set of tetracycline (Tc) riboswitch dimer sequences upstream of a GFP mRNA. The measured variable in this data set is the switching factor, which refers to the differential effect of the riboswitch in the presence or absence of Tc. The <bold>SARS-CoV-2 vaccine degradation</bold> data set (<xref rid="GR278870LIC20" ref-type="bibr">Leppek et al. 2022</xref>) encompasses a set of mRNA sequences that have been tuned for their structural features, stability, and translation efficiency. The average of the deg_Mg_50C values at each nucleotide is treated as the sequence-level target. Deg_Mg_50C has the highest correlation with other labels, including deg_pH10, deg_Mg_pH10, and deg_50C. The benchmarking data set also included a new data set generated by Sanofi encoding the hemagglutinin antigen for flu vaccines. Briefly, mRNA sequences, encoding the Influenza H3N2 A/Tasmania/503/2020 hemagglutinin protein, were tested for protein expression level in HeLa cells (Methods).</p><p>To compare CodonBERT's performance on these tasks, we have also applied several other state-of-the-art methods that have been previously used for mRNA property prediction with different model complexities, including TF-IDF (<xref rid="GR278870LIC38" ref-type="bibr">Rajaraman and Ullman 2011</xref>), TextCNN (<xref rid="GR278870LIC19" ref-type="bibr">Kim 2014</xref>), Codon2vec, RNABERT (<xref rid="GR278870LIC4" ref-type="bibr">Akiyama and Sakakibara 2022</xref>), and RNA-FM (<xref rid="GR278870LIC8" ref-type="bibr">Chen et al. 2022</xref>). <xref rid="GR278870LITB2" ref-type="table">Table 2</xref> presents the performance of CodonBERT and the other six methods on these downstream tasks. For each task, the first three rows are nucleotide-based methods (plain TextCNN, RNABERT, and RNA-FM), whereas the rest are codon-based methods (TF-IDF, plain TextCNN, Codon2vec, and CodonBERT). <xref rid="SD1" ref-type="supplementary-material">Supplemental Table S1</xref> provides complimentary loss values for these comparisons. Overall, we see that codon-based methods outperform nucleotide-based methods on most tasks. This is in part because of the critical role of codons on the protein expression. Moreover, the codon-based variant of TextCNN outperforms the original nucleotide implementation on most tasks.</p><table-wrap id="GR278870LITB2" position="float"><?disp-level 3?><label>Table 2.</label><caption><p>Comparison of CodonBERT to prior methods on seven downstream tasks</p></caption><table frame="hsides" rules="groups"><colgroup span="1"><col align="left" span="1"/><col align="char" char="." span="1"/><col align="char" char="." span="1"/><col align="char" char="." span="1"/><col align="char" char="." span="1"/><col align="char" char="." span="1"/><col align="char" char="." span="1"/><col align="char" char="." span="1"/></colgroup><thead><tr><th align="left" rowspan="1" colspan="1">Model</th><th align="center" rowspan="1" colspan="1">Flu vaccines</th><th align="center" rowspan="1" colspan="1">mRFP expression</th><th align="center" rowspan="1" colspan="1">Fungal expression</th><th align="center" rowspan="1" colspan="1"><italic>E. coli</italic> proteins</th><th align="center" rowspan="1" colspan="1">mRNA stability</th><th align="center" rowspan="1" colspan="1">Tc-riboswitch</th><th align="center" rowspan="1" colspan="1">SARS-CoV-2 vaccine degradation</th></tr></thead><tbody><tr><td rowspan="1" colspan="1">Nucleotide-based</td><td rowspan="1" colspan="1"/><td rowspan="1" colspan="1"/><td rowspan="1" colspan="1"/><td rowspan="1" colspan="1"/><td rowspan="1" colspan="1"/><td rowspan="1" colspan="1"/><td rowspan="1" colspan="1"/></tr><tr><td rowspan="1" colspan="1"> Plain TextCNN</td><td rowspan="1" colspan="1">0.72</td><td rowspan="1" colspan="1">0.62</td><td rowspan="1" colspan="1">0.53</td><td rowspan="1" colspan="1">0.39</td><td rowspan="1" colspan="1">0.01</td><td rowspan="1" colspan="1">0.41</td><td rowspan="1" colspan="1">0.55</td></tr><tr><td rowspan="1" colspan="1"> RNABERT<sub>+TextCNN</sub></td><td rowspan="1" colspan="1">0.65</td><td rowspan="1" colspan="1">0.40</td><td rowspan="1" colspan="1">0.41</td><td rowspan="1" colspan="1">0.39</td><td rowspan="1" colspan="1">0.16</td><td rowspan="1" colspan="1">0.47</td><td rowspan="1" colspan="1">0.64</td></tr><tr><td rowspan="1" colspan="1"> RNA-FM<sub>+TextCNN</sub></td><td rowspan="1" colspan="1">0.71</td><td rowspan="1" colspan="1">0.80</td><td rowspan="1" colspan="1">0.59</td><td rowspan="1" colspan="1">0.43</td><td rowspan="1" colspan="1">0.34</td><td rowspan="1" colspan="1">
<bold>0.58</bold>
</td><td rowspan="1" colspan="1">0.74</td></tr><tr><td rowspan="1" colspan="1">Codon-based</td><td rowspan="1" colspan="1"/><td rowspan="1" colspan="1"/><td rowspan="1" colspan="1"/><td rowspan="1" colspan="1"/><td rowspan="1" colspan="1"/><td rowspan="1" colspan="1"/><td rowspan="1" colspan="1"/></tr><tr><td rowspan="1" colspan="1"> TF-IDF</td><td rowspan="1" colspan="1">0.68</td><td rowspan="1" colspan="1">0.57</td><td rowspan="1" colspan="1">0.68</td><td rowspan="1" colspan="1">0.44</td><td rowspan="1" colspan="1">
<bold>0.54</bold>
</td><td rowspan="1" colspan="1">0.49</td><td rowspan="1" colspan="1">0.69</td></tr><tr><td rowspan="1" colspan="1"> Plain TextCNN</td><td rowspan="1" colspan="1">0.71</td><td rowspan="1" colspan="1">0.78</td><td rowspan="1" colspan="1">0.76</td><td rowspan="1" colspan="1">0.36</td><td rowspan="1" colspan="1">0.26</td><td rowspan="1" colspan="1">0.43</td><td rowspan="1" colspan="1">
<bold>0.80</bold>
</td></tr><tr><td rowspan="1" colspan="1"> Codon2vec<sub>+TextCNN</sub></td><td rowspan="1" colspan="1">0.72</td><td rowspan="1" colspan="1">0.77</td><td rowspan="1" colspan="1">0.61</td><td rowspan="1" colspan="1">0.43</td><td rowspan="1" colspan="1">0.33</td><td rowspan="1" colspan="1">0.56</td><td rowspan="1" colspan="1">0.70</td></tr><tr><td rowspan="1" colspan="1"> CodonBERT</td><td rowspan="1" colspan="1">
<bold>0.81</bold>
</td><td rowspan="1" colspan="1">
<bold>0.85</bold>
</td><td rowspan="1" colspan="1">
<bold>0.88</bold>
</td><td rowspan="1" colspan="1">
<bold>0.55</bold>
</td><td rowspan="1" colspan="1">0.51</td><td rowspan="1" colspan="1">0.56</td><td rowspan="1" colspan="1">0.77</td></tr></tbody></table><table-wrap-foot><fn id="fn3"><p>For regression tasks, the corresponding Spearman's rank correlation values are listed. For the classification task (<italic>E. coli</italic> protein data set), classification accuracy is calculated. The best values of correlation and accuracy for each task are in bold. The corresponding loss values are listed in <xref rid="SD1" ref-type="supplementary-material">Supplemental Table S1</xref>.</p></fn></table-wrap-foot></table-wrap><p>As for the detailed comparison, we observe that CodonBERT performed best on four of the seven tasks and second best (in most cases with very small difference) on two of the remaining three tasks. Plain codon-based TextCNN produced the best results for SARS-CoV-2 vaccine degradation, whereas it performed poorly on other prediction tasks including riboswitches, flu vaccines, and <italic>E. coli</italic>. The other two methods that were best performing for one of the data sets, for example, TF-IDF and RNA-FM, did not perform well on the other tasks.</p><p>Both secondary structure and codon usage play critical roles in mRNA vaccine expression (<xref rid="GR278870LIC24" ref-type="bibr">Mauger et al. 2019</xref>; <xref rid="GR278870LIC1" ref-type="bibr">Agarwal and Kelley 2022</xref>). Stable secondary structure increases mRNA stability in solution (<xref rid="GR278870LIC24" ref-type="bibr">Mauger et al. 2019</xref>), and optimal codons improve cellular mRNA stability (<xref rid="GR278870LIC1" ref-type="bibr">Agarwal and Kelley 2022</xref>). Therefore, mRNA stability, SARS-CoV-2 vaccine degradation, and Tc-riboswitch data sets are strongly affected by local and global secondary structure patterns encoded on top of RNA sequences. Although CodonBERT is a codon-based model, it outperforms RNABERT and RNA-FM, which were demonstrated to capture rich structural information from large-scale noncoding RNAs. This may indicate that CodonBERT also learns coevolutionary information and structural properties from millions of mRNA sequences.</p><p>Nucleotide embeddings learned from noncoding RNAs, for example, RNA-FM<sub>+TextCNN</sub>, leads to significantly better results than plain nucleotide-based TextCNN on most tasks. This may indicate that structural information, even from noncoding RNA sequences, is beneficial to solving mRNA translation and stability problems. Although both RNABERT and RNA-FM are pretrained BERT models from noncoding RNA sequences, their performance differs. This may be attributed to the training data size and model capacity of RNA-FM, which is significantly larger than that of RNABERT.</p></sec></sec><sec id="s2" disp-level="1"><title>Discussion</title><p>To enable the analysis and prediction of mRNA properties, we utilized 10 million mRNA CDSs from several species to train a LLM (CodonBERT) and to establish a foundational model. Our primary focus is on optimizing mRNA vaccines and drugs, concentrating specifically on sequences pertinent to these applications, including those from host cells and viruses critical for vaccine development.</p><p>The model optimizes two self-supervised tasks: codon completion and taxonomic identification. Like other unsupervised LLMs, we expected that such a foundational model will learn to capture aspects of natural selection that favor mRNA sequences with high expression and stable structure. Analysis of the resulting model indicates that it indeed learns several relevant biological properties for codons and sequences.</p><p>Projection of codon embedding obtained from CodonBERT produces distinct clusters that adhere to the amino acid types. Besides, the analysis of alanine codons revealed notable clustering patterns: “GCU” and “GCC” cluster separately from “GCA” and “GCG.” Cosine similarity calculations (<xref rid="SD1" ref-type="supplementary-material">Supplemental Fig. S3</xref>) supported this grouping, showing higher similarities between “GCG” and the nonsynonymous codons “CCG” (proline), “UCG” (serine), and “ACG” (threonine) than with its synonymous counterparts. This unexpected pattern, consistent with our projection plot findings, presents an interesting anomaly in codon behavior. The reasons for these unusual similarities remain unclear, suggesting an area for further exploration that could provide new insights into codon usage and gene expression mechanisms. In-depth analysis of CodonBERT representation of a set of genes from different organisms revealed that CodonBERT autonomously learns the genetic code and principles of evolutionary taxonomy.</p><p>We also utilized CodonBERT to perform several supervised prediction tasks for mRNA properties. These include data sets testing for recombinant protein expression, mRNA degradation, mRNA stability, and more. Our results indicate that CodonBERT is the top-performing method overall and ranks first or second in performance for six of the seven tasks. All other methods we compared performed poorly on all, or some of the tasks. Thus, for a new task, the use of CodonBERT is likely to lead to either the best or close to best results. CodonBERT's success in the more structurally related tasks (including mRNA stability and the TC riboswitch data sets) indicates that it can learn coevolutionary and structural concepts using large-scale mRNA sequences.</p><p>For the vaccine-related downstream tasks, CodonBERT generally exhibited robust performance. It was 10% better than the second-best method for the new hemagglutinin flu vaccine expression data set and very close (2% difference) to the top-performing model for the SARS-CoV-2 vaccine degradation data set. Although fungal sequences are not included in pretraining sequences, the CodonBERT model shows its generalization to a downstream fungal data set. RNABERT and RNABERT are also pretrained RNA LLMs; however, they are nucleotide based and trained on noncoding RNAs. Because they do not explicitly capture codon usage, these methods are inferior to CodonBERT on the protein expression-related tasks.</p><p>The flu vaccine data set uses N1-methylpseudouridine in RNA modification for bypassing the innate immune response and enhancing protein synthesis from mRNA. CodonBERT model's ability to adapt to these modifications without prior exposure to similar natural data exemplifies its robustness and versatility.</p><p>The one exception in terms of performance was observed for the mRNA stability task. Stability is known to be structure dependent, and stable structures such as stem-loops or hairpin structures can impede degradation enzymes, protecting the mRNA from rapid decay. A possible reason for the reduction in performance for this data set is that structural properties are highly dependent on nucleotides, whereas CodonBERT is a codon-based model. One possible solution for this is a model that combines codon and nucleotide representation. Similarly, mRNA modification events including capping at the 5′ end and polyadenylation at the 3′ end in eukaryotes are not currently encoded in our model but can also impact mRNA stability. However, extending CodonBERT to include UTRs will be an important direction for future work.</p><p>Although the tasks we have focused on are supervised in nature, CodonBERT, as an LLM, can be utilized for generative purposes. Specifically, we envision using this model for codon optimization of various heterologous proteins for vaccines. Other generative tasks can include sampling mRNA for synthetic biology use cases (e.g., creation of optimized mRNA constructs for genome editing).</p><p>To conclude, our findings suggest that CodonBERT could serve as a versatile and foundational model for the development of new mRNA-based vaccines and the engineering and recombinant production of industrial and therapeutic proteins.</p></sec><sec id="s3" disp-level="1"><title>Methods</title><sec id="s3a" disp-level="2"><title>Assembly of mRNA sequences for pretraining</title><p>We collected mRNA sequences across diverse organisms for pretraining from NCBI (<xref rid="GR278870LIC47" ref-type="bibr">Wheeler et al. 2007</xref>). The data sets included <italic>mammalian</italic> reference sequences (<ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://www.ncbi.nlm.nih.gov/datasets/taxonomy/40674/" ext-link-type="uri">https://www.ncbi.nlm.nih.gov/datasets/taxonomy/40674/</ext-link>), <italic>bacteria</italic> (<italic>E. coli</italic>) reference sequences (<ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://www.ncbi.nlm.nih.gov/datasets/taxonomy/562/" ext-link-type="uri">https://www.ncbi.nlm.nih.gov/datasets/taxonomy/562/</ext-link>), <italic>Homo sapiens virus</italic> complete nucleotides (<ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://www.ncbi.nlm.nih.gov/labs/virus/vssi/#/" ext-link-type="uri">https://www.ncbi.nlm.nih.gov/labs/virus/vssi/#/</ext-link>), and <italic>yeast</italic> strains (<ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="http://sgd-archive.yeastgenome.org/sequence/strains/" ext-link-type="uri">http://sgd-archive.yeastgenome.org/sequence/strains/</ext-link>). Each sequence is with a label representing its taxonomic group.</p><p>We preprocessed all the sequences and filtered out some invalid and replicate ones by requiring the mRNA sequences with the sequence length multiples of three, starting with the start codon (“AUG”) and ending with stop codons (“UAA,” “UAG,” or “UGA”), and only including nucleotides from the set {A, U, G, C, N} (replacing T with U). After preprocessing, 10 million mRNA sequences were valid.</p><p>To build sequence pairs for the homologous sequence prediction task, 50% of the sequence pairs consisted of two sequences belonging to one of the 14 categories. The remaining 50% of sequence pairs included two sequences that were randomly sampled from two different categories.</p></sec><sec id="s3b" disp-level="2"><title>Model architecture</title><p>A codon is composed of three adjacent nucleotides. There are five different options for each of these three positions {A, U, G, C, N}, leading to a total of 5<sup>3</sup> (125) possible combinations. Additionally, five special tokens are added to the vocabulary: classifier token ([CLS]), separator token ([SEP]), unknown token ([UNK]), padding token ([PAD]), and masking token ([MASK]). Thus, in total, there are 130 tokens in the vocabulary of CodonBERT.</p><p>As shown in <xref rid="GR278870LIF1" ref-type="fig">Figure 1</xref>B, CodonBERT takes a sequence pair as input and concatenates them using a separator token ([SEP]). It then adds a classifier token ([CLS]) and a separator token ([SEP]) at the beginning and end of the combined sequence, respectively. CodonBERT constructs the input embedding by concatenating codon, position, and segment embeddings. Absolute positions are utilized with values initialized from one to <italic>n</italic><sub>1</sub> + <italic>n</italic><sub>2</sub> + 3 along the concatenated sequence, where <italic>n</italic><sub>1</sub> and <italic>n</italic><sub>2</sub> are the codon-wise length of two sequences plus three specially added tokens ([CLS] and [SPE]). The segment value is either one or two to distinguish two sequences. These three types of embedding matrices are learned across 10 millions of mRNA sequences.</p><p>The combined input embedding is fed into the CodonBERT model, which consists of a stack of 12 layers of bidirectional transformer encoders (<xref rid="GR278870LIC44" ref-type="bibr">Vaswani et al. 2017</xref>) as shown in <xref rid="GR278870LIF1" ref-type="fig">Figure 1</xref>C. Each transformer layer processes its input using 12 self-attention heads and then outputs a representation for each position with hidden size 768. In each layer, the multihead self-attention mechanism captures the contextual information of the input sequence by considering all the other codons in the sequence. A key benefit of self-attention mechanism is the connection learned between all pairs of positions in an input sequence using parallel computation, which enables CodonBERT to model not only short-range but also long-range interactions, which impact translation efficiency and stability (<xref rid="GR278870LIC6" ref-type="bibr">Aw et al. 2016</xref>). Next a feed-forward neural network is added to apply a nonlinear transformation to the output hidden representation from the self-attention network. A residual connection is employed around each of the multihead attention and feed-forward networks. After processing the input sequence with a stack of transformer encoders, CodonBERT produces the final contextualized codon representations, which is followed by a classification layer to produce probability distribution over the vocabulary during pretraining.</p></sec><sec id="s3c" disp-level="2"><title>Pretraining CodonBERT</title><p>Model architecture of CodonBERT and the training for the two tasks are illustrated in <xref rid="GR278870LIF1" ref-type="fig">Figure 1</xref>B. Prior to being fed into the model, the input mRNA sequence is first tokenized into a list of codons. Next, a fraction of the input codons (15%) is randomly selected and replaced by the masking token ([MASK]). The self-training loop optimizes CodonBERT to predict the masked codons based on the remaining ones, taking into account interactions between the missing and unmasked codons. A probability distribution over 64 possible codons is produced by CodonBERT for the masked positions. The average cross entropy loss ℒ<sub>MLM</sub> over the masked positions <italic>M</italic> is calculated by the optimization function:
</p><disp-formula id="GR278870LIUM1"><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="UM1" display="block" overflow="scroll"><mml:mrow><mml:msub><mml:mrow><mml:mi class="MJX-tex-caligraphic" mathvariant="script">L</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">MLM</mml:mi></mml:mrow></mml:mrow></mml:msub></mml:mrow><mml:mo>=</mml:mo><mml:mo>−</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="double-struck">E</mml:mi></mml:mrow></mml:mrow></mml:mrow><mml:mrow><mml:mi>x</mml:mi><mml:mo>∼</mml:mo><mml:mi>X</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="double-struck">E</mml:mi></mml:mrow></mml:mrow></mml:mrow><mml:mi>M</mml:mi></mml:msub></mml:mrow><mml:munder><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>∈</mml:mo><mml:mi>M</mml:mi></mml:mrow></mml:munder><mml:mrow><mml:mi>log</mml:mi><mml:mo>⁡</mml:mo><mml:mi>p</mml:mi><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo fence="false">|</mml:mo><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mi>M</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:math></disp-formula><p>
where <italic>X</italic> represents a batch of sequences, <italic>x</italic> is one sequence, and <italic>x</italic><sub><italic>i</italic></sub> is the original codon for the position <italic>i</italic>. <italic>x</italic><sub><italic>M</italic></sub> is the masked input with a set of positions <italic>M</italic> masked. (<italic>x</italic><sub><italic>i</italic></sub>∣<italic>x</italic><sub><italic>M</italic></sub>) indicates the output probability of the real codon <italic>x</italic><sub><italic>i</italic></sub> given all the remaining codons in the masked sequence <italic>x</italic><sub><italic>M</italic></sub>.</p><p>For the STP task, the output embedding of the classifier token ([CLS]) is used for prediction about whether these two sequences belong to the same class (binary classification). The average cross entropy loss ℒ<sub>STP</sub> is computed as
</p><disp-formula id="GR278870LIUM2"><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="UM2" display="block" overflow="scroll"><mml:mrow><mml:msub><mml:mrow><mml:mi class="MJX-tex-caligraphic" mathvariant="script">L</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">STP</mml:mi></mml:mrow></mml:mrow></mml:msub></mml:mrow><mml:mo>=</mml:mo><mml:mo>−</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="double-struck">E</mml:mi></mml:mrow></mml:mrow></mml:mrow><mml:mi>N</mml:mi></mml:msub></mml:mrow><mml:munderover><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>N</mml:mi></mml:munderover><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mi>y</mml:mi><mml:mi>n</mml:mi></mml:msub></mml:mrow><mml:mi>log</mml:mi><mml:mo>⁡</mml:mo><mml:mrow><mml:mspace width="0.4em"/><mml:msub><mml:mi>p</mml:mi><mml:mi>n</mml:mi></mml:msub></mml:mrow></mml:mrow><mml:mo>+</mml:mo><mml:mo>(</mml:mo><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:mrow><mml:msub><mml:mi>y</mml:mi><mml:mi>n</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo><mml:mi>log</mml:mi><mml:mo>⁡</mml:mo><mml:mo>(</mml:mo><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:mrow><mml:mspace width="0.4em"/><mml:msub><mml:mi>p</mml:mi><mml:mi>n</mml:mi></mml:msub></mml:mrow><mml:mo>)</mml:mo><mml:mo>]</mml:mo><mml:mo>,</mml:mo></mml:math></disp-formula><p>
where <italic>N</italic> represents the number of sequence pairs; <italic>y</italic><sub><italic>n</italic></sub> is the expected value, which is one when two sequences are from the same taxonomic group and zero when they are not; and <italic>p</italic><sub><italic>n</italic></sub> indicates the predicted probability of two sequences belonging to the same category. The total loss is the sum of the losses from both tasks (<inline-formula id="il1"><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IL1" display="inline" overflow="scroll"><mml:mrow><mml:msub><mml:mrow><mml:mi class="MJX-tex-caligraphic" mathvariant="script">L</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">MLM</mml:mi></mml:mrow></mml:mrow></mml:msub></mml:mrow><mml:mo>+</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi class="MJX-tex-caligraphic" mathvariant="script">L</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mi mathvariant="normal">STP</mml:mi></mml:mrow></mml:mrow></mml:msub></mml:mrow></mml:math></inline-formula>).</p><p>We used a batch size of 128 with a sequence length limit of 1024 and trained the model around seven epochs in 2 weeks. Because the inputs of CodonBERT are sequence pairs, the length of each sequence is limited up to 512 codons; therefore, the length of the combined sequence is less than 1024. Sequences exceeding the length limitation are split into fragments no longer than 512. Pretraining CodonBERT using 10 million mRNA sequences on 4 A10G GPUs with 96 GB GPU memory and 192 GB memory took ∼2 weeks. The model was configured with the following specifications: a sequence length of 1024, as well as 12 layers, each with 12 attention heads. The model's hidden size is set at 768. Overall, the model encompasses about 110 million parameters.</p><p>CodonBERT was also applied to a wide range of downstream tasks. For this, we can use either a single or a pair of sequences as input (<xref rid="GR278870LIF1" ref-type="fig">Figs. 1</xref>B, <xref rid="GR278870LIF3" ref-type="fig">3</xref>C). To perform supervised analysis, the output embedding is followed by an output layer that is trained for the specific task (protein expression level prediction, mRNA stability, etc.). We also conducted an ablation study in which we pretrained a model with the same architecture using only the MLM task. Although performance was good, it still did not match performance of the model trained with both types of metrics. See <xref rid="SD1" ref-type="supplementary-material">Supplemental Figure S5</xref> and <xref rid="SD1" ref-type="supplementary-material">Supplemental Table S2</xref>.</p><fig id="GR278870LIF3" position="float"><?disp-level 3?><label>Figure 3.</label><caption><p>Comparison to prior methods (TF-IDF, Codon2vec, RNABERT, and RNA-FM) and fine-tuning CodonBERT on downstream data sets. (<italic>A</italic>) Given an input corpus with <italic>m</italic> mRNA sequences, TF-IDF is used to construct a feature matrix followed by a random forest regression model. (<italic>B</italic>) Use a TextCNN model to learn task-specific nucleotide or codon representations. The model is able to fine-tune pretrained representations by initializing the embedding layers with stacked codon or nucleotide embeddings extracted from pretrained language models (Codon2vec, RNABERT, and RNA-FM). <italic>n</italic> is the number of codons in the input sequence, and <italic>d</italic> is the dimension of the token embedding. As a baseline, plain TextCNN initializes the embedding layer with a standard normal distribution. (<italic>C</italic>) Fine-tune the pretrained CodonBERT model on a given downstream task directly by keeping all the parameters trainable.</p></caption><alternatives><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="image" xlink:href="1027f03.jpg"><?cloudpmc-path blobs/d27d/11368176/55221e0a5546/1027f03.jpg?><?cloudpmc-bucket cdn?><?image-server-status LOAD_COMPLETED?><?original-height 969?><?original-width 1975?><?scaled-height 388?><?scaled-width 790?></graphic><graphic xmlns:xlink="http://www.w3.org/1999/xlink" content-type="thumb" xlink:href="1027f03.gif"><?cloudpmc-path blobs/d27d/11368176/7bd37247d15e/1027f03.gif?><?cloudpmc-bucket cdn?></graphic></alternatives></fig></sec><sec id="s3d" disp-level="2"><title>Pretraining Codon2vec</title><p>Word2vec is a popular neural network-based model that is used to learn distributed representations of words in a corpus (<xref rid="GR278870LIC29" ref-type="bibr">Mikolov et al. 2013</xref>). Like the LLM mentioned above, Word2vec also learns token representation from a large-scale text corpus, and the embedding from both methods can be utilized as input features for downstream tasks. Unlike LLMs, it only produces a single assignment for each token, which usually correlates with the maximum or most frequent context around the token in the corpus.</p><p>An existing work has applied Word2vec to the fungal genomes, studied codon usage bias, and built a predictive model for gene expression (<xref rid="GR278870LIC48" ref-type="bibr">Wint et al. 2022</xref>). However, there is no application on a large-scale mRNA sequence data set. Therefore, for comparison, we trained our own Codon2vec model on the collected mRNA sequences. Using the Gensim library (<xref rid="GR278870LIC39" ref-type="bibr">Řehůřek and Sojka 2010</xref>), the Codon2vec model opted for a skip-gram architecture, accompanied by a window size of five and a minimum count threshold of 10 codons. The model was trained using hierarchical SoftMax and negative sampling methodologies.</p><p>The input sequences for pretraining Codon2vec were processed through a filtration system that selected sequences containing fewer than 1000 nt. This filtration stage was necessitated by the constrained model capacity of the Word2vec neural network. Upon completion of this process, we retrieved a total of around 2 million sequences, which were subsequently subjected to tokenization into <italic>k</italic>-mers. These <italic>k</italic>-mers serve as representations of all corresponding codons.</p></sec><sec id="s3e" disp-level="2"><title>In vitro transcription, cell culture, and transfections</title><p>As shown in <xref rid="SD1" ref-type="supplementary-material">Supplemental Figure S4</xref>, mRNA sequences were designed using to encode the Influenza H3N2 A/Tasmania/503/2020 hemagglutinin protein. Sequences corresponding to these candidates were synthesized as gene fragments and PCR-amplified to generate template DNA for high-throughput in vitro transcription reaction containing N1-methylpseudouridine. The resulting purified precursor mRNA was reacted further via enzymatic addition of a 5′ cap structure (Cap 1) and a 3′ poly(A) tail of ∼200 nt in length as determined by capillary electrophoresis.</p><p>HeLa cells were used to evaluate the expression of the protein encoded by different mRNA sequences. Cells were cultured and maintained in MEM (Corning) containing 10% (v/v) heat-inactivated FBS (Gibco). To evaluate the expression of candidate mRNAs, HeLa cells were transiently transfected with mRNAs complexed with Lipofectamine MessengerMax (Thermo Fisher Scientific). Unknown mRNAs were thawed, diluted in Opti-MEM, combined with Lipofectamine for 10 min, and then further diluted in Opti-MEM. To prepare the cells for reverse transfection, HeLa cells were collected from culture flasks using TrypLE and were diluted in complete growth medium such that each well will be seeded with 2E4 live cells. Complexed mRNAs (20 ng/well) were added to triplicate wells of a 96-well poly-D-lysine PhenoPlate (PerkinElmer) and were combined with 2E4 HeLa cells. Plates were rested at RT briefly before incubation in a tissue culture incubator for 20 h + 30 min. At the endpoint, cells were lysed in RIPA (Thermo Fisher Scientific) supplemented with OmniCleave (Lucigen) and HALT protease inhibitor (Thermo Fisher Scientific). The hemagglutinin expression in cell lysates was determined using a quantitative sandwich ELISA, and the expression level of unknown mRNAs was normalized to the value from a known benchmark mRNA sequence.</p></sec><sec id="s3f" disp-level="2"><title>Comparisons to other methods</title><p>We compared CodonBERT to several prior methods that have been used to model and analyze RNA sequences:
</p><list list-type="bullet"><list-item><p>Term frequency-inverse document frequency (TF-IDF) (<xref rid="GR278870LIC38" ref-type="bibr">Rajaraman and Ullman 2011</xref>) is a numerical statistic that is commonly used as a weighting scheme in information retrieval and natural language processing. In the context of mRNA sequences, TF-IDF is applied to measure the significance of each codon in a sequence. A high TF-IDF value of a codon indicates that the codon is important in a particular mRNA sequence and is rare across all mRNA sequences in the corpus. <xref rid="GR278870LIF3" ref-type="fig">Figure 3</xref>A illustrates the application TF-IDF in the benchmark.</p></list-item><list-item><p>Convolutional neural network (CNN) was first proposed and has been commonly used in image recognition and was later also applied for text analysis TextCNN (<xref rid="GR278870LIC19" ref-type="bibr">Kim 2014</xref>). TextCNN consists of multiple types of layers, including an embedding layer, convolutional layer, pooling layer, and fully connected layer, as shown in <xref rid="GR278870LIF3" ref-type="fig">Figure 3</xref>B. Each row of the embedding layer represents a token.</p></list-item><list-item><p>RNABERT (<xref rid="GR278870LIC4" ref-type="bibr">Akiyama and Sakakibara 2022</xref>) and RNA-FM (<xref rid="GR278870LIC8" ref-type="bibr">Chen et al. 2022</xref>) are RNA LLMs. However, they are pretrained on noncoding RNAs to learn and encode structural and functional properties in the output nucleotide embedding.</p></list-item></list></sec><sec id="s3g" disp-level="2"><title>Software availability</title><p>Software and data are available in the Sanofi GitHub (<ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://github.com/Sanofi-Public/CodonBert" ext-link-type="uri">https://github.com/Sanofi-Public/CodonBert</ext-link>) and as <xref rid="SD2" ref-type="supplementary-material">Supplemental Code</xref>.</p></sec></sec><sec id="SM1" disp-level="1"><title>Supplementary Material</title><supplementary-material id="SD1" position="float"><?disp-level 2?><label>Supplement 1</label><media xmlns:xlink="http://www.w3.org/1999/xlink" id="d67e1358" xlink:href="Supplemental_Material.zip" mimetype="application" mime-subtype="zip"><?cloudpmc-path d27d/11368176/365fed031c19/Supplemental_Material.zip?><?cloudpmc-bucket app?><?size 4754142?></media></supplementary-material><supplementary-material id="SD2" position="float"><?disp-level 2?><label>Supplement 2</label><media xmlns:xlink="http://www.w3.org/1999/xlink" id="d67e1361" xlink:href="Supplemental_Source_Code.zip" mimetype="application" mime-subtype="zip"><?cloudpmc-path d27d/11368176/6a97e4d854b0/Supplemental_Source_Code.zip?><?cloudpmc-bucket app?><?size 192998894?></media></supplementary-material></sec><sec id="ack1" sec-type="ack" disp-level="1"><title>Acknowledgments</title><p>We thank Eleni Litsa for support in preparing mRNA designs.</p><p><italic>Author contributions</italic>: Z.B-J. and S.J. proposed this idea. S.L. and S.M. implemented the algorithm and performed data collection and the in silico experiments. S.L., S.M., S.R., and S.J. wrote the paper. R.L., M.B., M.M., J.M., D.Z., J.W., A.B., F.U.M., V.A., and Z.B-J. revised the manuscript. L.K-A. developed the public code repository. M.M, J.M., F.P., D.Z., J.W., A.B., K.T., M.Z., M.W., X.G., R.C., C.A., J.S., L.B., S.C., A.D., T.S., F.U.M., and V.A. built the flu vaccine data set.</p></sec><sec id="fn-group1" sec-type="fn-group" disp-level="1"><title>Footnotes</title><fn-group><fn id="fn4"><p>[Supplemental material is available for this article.]</p></fn><fn id="fn5"><p>Article published online before print. Article, supplemental material, and publication date are at <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://www.genome.org/cgi/doi/10.1101/gr.278870.123" ext-link-type="uri">https://www.genome.org/cgi/doi/10.1101/gr.278870.123</ext-link>.</p></fn></fn-group></sec><sec id="s5" disp-level="1"><title>Competing interest statement</title><p>All authors are Sanofi employees and may hold shares and/or stock options in the company.</p></sec><sec id="ref-list1" sec-type="ref-list" disp-level="1"><title>References</title><sec id="ref-list1_sec2" disp-level="2"><ref-list><ref id="GR278870LIC1"><mixed-citation><named-content content-type="citation-string">Agarwal V, Kelley DR. 2022. The genetic and biochemical determinants of mRNA degradation rates in mammals. Genome Biol
23: 245. 10.1186/s13059-022-02811-x
</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1186/s13059-022-02811-x"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC9684954"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="36419176"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Genome Biol&amp;title=The genetic and biochemical determinants of mRNA degradation rates in mammals&amp;author=V Agarwal&amp;author=DR Kelley&amp;volume=23&amp;publication_year=2022&amp;pages=245&amp;pmid=36419176&amp;doi=10.1186/s13059-022-02811-x&amp;"/></mixed-citation></ref><ref id="GR278870LIC2"><mixed-citation><named-content content-type="citation-string">Agarwal V, Shendure J. 2020. Predicting mRNA abundance directly from genomic sequence using deep convolutional neural networks. Cell Rep
31: 107663. 10.1016/j.celrep.2020.107663
</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1016/j.celrep.2020.107663"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="32433972"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Cell Rep&amp;title=Predicting mRNA abundance directly from genomic sequence using deep convolutional neural networks&amp;author=V Agarwal&amp;author=J Shendure&amp;volume=31&amp;publication_year=2020&amp;pages=107663&amp;pmid=32433972&amp;doi=10.1016/j.celrep.2020.107663&amp;"/></mixed-citation></ref><ref id="GR278870LIC3"><mixed-citation><named-content content-type="citation-string">Ahmad HI, Jabbar A, Mushtaq N, Javed Z, Hayyat MU, Bashir J, Naseeb I, Abideen ZU, Ahmad N, Chen J. 2022. Immune tolerance vs. immune resistance: the interaction between host and pathogens in infectious diseases. Front Vet Sci
9: 827407. 10.3389/fvets.2022.827407
</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.3389/fvets.2022.827407"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC9001959"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="35425833"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Front Vet Sci&amp;title=Immune tolerance vs. immune resistance: the interaction between host and pathogens in infectious diseases&amp;author=HI Ahmad&amp;author=A Jabbar&amp;author=N Mushtaq&amp;author=Z Javed&amp;author=MU Hayyat&amp;volume=9&amp;publication_year=2022&amp;pages=827407&amp;pmid=35425833&amp;doi=10.3389/fvets.2022.827407&amp;"/></mixed-citation></ref><ref id="GR278870LIC4"><mixed-citation><named-content content-type="citation-string">Akiyama M, Sakakibara Y. 2022. Informative RNA base embedding for RNA structural alignment and clustering by deep representation learning. NAR Genom Bioinform
4: lqac012. 10.1093/nargab/lqac012
</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1093/nargab/lqac012"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC8862729"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="35211670"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=NAR Genom Bioinform&amp;title=Informative RNA base embedding for RNA structural alignment and clustering by deep representation learning&amp;author=M Akiyama&amp;author=Y Sakakibara&amp;volume=4&amp;publication_year=2022&amp;pages=lqac012&amp;pmid=35211670&amp;doi=10.1093/nargab/lqac012&amp;"/></mixed-citation></ref><ref id="GR278870LIC5"><mixed-citation><named-content content-type="citation-string">Al-Hawash AB, Zhang X, Ma F. 2017. Strategies of codon optimization for high-level heterologous protein expression in microbial expression systems. Gene Rep
9: 46–53. 10.1016/j.genrep.2017.08.006</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1016/j.genrep.2017.08.006"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Gene Rep&amp;title=Strategies of codon optimization for high-level heterologous protein expression in microbial expression systems&amp;author=AB Al-Hawash&amp;author=X Zhang&amp;author=F Ma&amp;volume=9&amp;publication_year=2017&amp;pages=46-53&amp;doi=10.1016/j.genrep.2017.08.006&amp;"/></mixed-citation></ref><ref id="GR278870LIC6"><mixed-citation><named-content content-type="citation-string">Aw JGA, Shen Y, Wilm A, Sun M, Lim XN, Boon K-L, Tapsin S, Chan Y-S, Tan C-P, Sim AY, et al. 
2016. In vivo mapping of eukaryotic RNA interactomes reveals principles of higher-order organization and regulation. Mol Cell
62: 603–617. 10.1016/j.molcel.2016.04.028
</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1016/j.molcel.2016.04.028"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="27184079"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Mol Cell&amp;title=In vivo mapping of eukaryotic RNA interactomes reveals principles of higher-order organization and regulation&amp;author=JGA Aw&amp;author=Y Shen&amp;author=A Wilm&amp;author=M Sun&amp;author=XN Lim&amp;volume=62&amp;publication_year=2016&amp;pages=603-617&amp;pmid=27184079&amp;doi=10.1016/j.molcel.2016.04.028&amp;"/></mixed-citation></ref><ref id="GR278870LIC7"><mixed-citation><named-content content-type="citation-string">Bepler T, Berger B. 2021. Learning the protein language: evolution, structure, and function. Cell Syst
12: 654–669.e3. 10.1016/j.cels.2021.05.017
</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1016/j.cels.2021.05.017"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC8238390"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="34139171"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Cell Syst&amp;title=Learning the protein language: evolution, structure, and function&amp;author=T Bepler&amp;author=B Berger&amp;volume=12&amp;publication_year=2021&amp;pages=654-669.e3&amp;pmid=34139171&amp;doi=10.1016/j.cels.2021.05.017&amp;"/></mixed-citation></ref><ref id="GR278870LIC8"><mixed-citation><named-content content-type="citation-string">Chen J, Hu Z, Sun S, Tan Q, Wang Y, Yu Q, Zong L, Hong L, Xiao J, Shen T, et al. 
2022. Interpretable RNA foundation model from unannotated data for highly accurate RNA structure and function predictions. bioRxiv
10.1101/2022.08.06.503062</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1101/2022.08.06.503062"/></mixed-citation></ref><ref id="GR278870LIC9"><mixed-citation><named-content content-type="citation-string">Devlin J, Chang M-W, Lee K, Toutanova K. 2019. BERT: pre-training of deep bidirectional transformers for language understanding. In <italic>Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1</italic> (<italic>Long and Short Papers</italic>) (ed. Burstein J, et al.), pp. 4171–4186. Association for Computational Linguistics, Minneapolis. <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://aclanthology.org/N19-1423" ext-link-type="uri">https://aclanthology.org/N19-1423</ext-link>. 10.18653/v1/N19-1423</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.18653/v1/N19-1423"/></mixed-citation></ref><ref id="GR278870LIC10"><mixed-citation><named-content content-type="citation-string">Diez M, Medina-Muñoz SG, Castellano LA, da Silva Pescador G, Wu Q, Bazzini AA. 2022. iCodon customizes gene expression based on the codon composition. Sci Rep
12: 12126. 10.1038/s41598-022-15526-7
</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1038/s41598-022-15526-7"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC9287306"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="35840631"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Sci Rep&amp;title=iCodon customizes gene expression based on the codon composition&amp;author=M Diez&amp;author=SG Medina-Muñoz&amp;author=LA Castellano&amp;author=G da Silva Pescador&amp;author=Q Wu&amp;volume=12&amp;publication_year=2022&amp;pages=12126&amp;pmid=35840631&amp;doi=10.1038/s41598-022-15526-7&amp;"/></mixed-citation></ref><ref id="GR278870LIC11"><mixed-citation><named-content content-type="citation-string">Ding Z, Guan F, Xu G, Wang Y, Yan Y, Zhang W, Wu N, Yao B, Huang H, Tuller T, et al. 
2022. MPEPE, a predictive approach to improve protein expression in E. coli based on deep learning. Comput Struct Biotechnol J
20: 1142–1153. 10.1016/j.csbj.2022.02.030
</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1016/j.csbj.2022.02.030"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC8913310"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="35317239"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Comput Struct Biotechnol J&amp;title=MPEPE, a predictive approach to improve protein expression in E. coli based on deep learning&amp;author=Z Ding&amp;author=F Guan&amp;author=G Xu&amp;author=Y Wang&amp;author=Y Yan&amp;volume=20&amp;publication_year=2022&amp;pages=1142-1153&amp;pmid=35317239&amp;doi=10.1016/j.csbj.2022.02.030&amp;"/></mixed-citation></ref><ref id="GR278870LIC12"><mixed-citation><named-content content-type="citation-string">Grigoriev IV, Nikitin R, Haridas S, Kuo A, Ohm R, Otillar R, Riley R, Salamov A, Zhao X, Korzeniewski F, et al. 
2014. MycoCosm portal: gearing up for 1000 fungal genomes. Nucleic Acids Res
42: D699–D704. 10.1093/nar/gkt1183
</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1093/nar/gkt1183"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC3965089"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="24297253"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nucleic Acids Res&amp;title=MycoCosm portal: gearing up for 1000 fungal genomes&amp;author=IV Grigoriev&amp;author=R Nikitin&amp;author=S Haridas&amp;author=A Kuo&amp;author=R Ohm&amp;volume=42&amp;publication_year=2014&amp;pages=D699-D704&amp;pmid=24297253&amp;doi=10.1093/nar/gkt1183&amp;"/></mixed-citation></ref><ref id="GR278870LIC13"><mixed-citation><named-content content-type="citation-string">Groher F, Bofill-Bosch C, Schneider C, Braun J, Jager S, Geißler K, Hamacher K, Suess B. 2018. Riboswitching with ciprofloxacin: development and characterization of a novel RNA regulator. Nucleic Acids Res
46: 2121–2132. 10.1093/nar/gkx1319
</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1093/nar/gkx1319"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC5829644"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="29346617"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nucleic Acids Res&amp;title=Riboswitching with ciprofloxacin: development and characterization of a novel RNA regulator&amp;author=F Groher&amp;author=C Bofill-Bosch&amp;author=C Schneider&amp;author=J Braun&amp;author=S Jager&amp;volume=46&amp;publication_year=2018&amp;pages=2121-2132&amp;pmid=29346617&amp;doi=10.1093/nar/gkx1319&amp;"/></mixed-citation></ref><ref id="GR278870LIC14"><mixed-citation><named-content content-type="citation-string">Groher A-C, Jager S, Schneider C, Groher F, Hamacher K, Suess B. 2019. Tuning the performance of synthetic riboswitches using machine learning. ACS Synth Biol
8: 34–44. 10.1021/acssynbio.8b00207
</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1021/acssynbio.8b00207"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="30513199"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=ACS Synth Biol&amp;title=Tuning the performance of synthetic riboswitches using machine learning&amp;author=A-C Groher&amp;author=S Jager&amp;author=C Schneider&amp;author=F Groher&amp;author=K Hamacher&amp;volume=8&amp;publication_year=2019&amp;pages=34-44&amp;pmid=30513199&amp;doi=10.1021/acssynbio.8b00207&amp;"/></mixed-citation></ref><ref id="GR278870LIC15"><mixed-citation><named-content content-type="citation-string">Hallee L, Rafailidis N, Gleghorn JP. 2023. cdsBERT: extending protein language models with codon awareness. bioRxiv
10.1101/2023.09.15.558027</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1101/2023.09.15.558027"/></mixed-citation></ref><ref id="GR278870LIC16"><mixed-citation><named-content content-type="citation-string">Jackson LA, Anderson EJ, Rouphael NG, Roberts PC, Makhene M, Coler RN, McCullough MP, Chappell JD, Denison MR, Stevens LJ, et al. 
2020a. An mRNA vaccine against SARS-CoV-2: preliminary report. N Engl J Med
383: 1920–1931. 10.1056/NEJMoa2022483
</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1056/NEJMoa2022483"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC7377258"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="32663912"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=N Engl J Med&amp;title=An mRNA vaccine against SARS-CoV-2: preliminary report&amp;author=LA Jackson&amp;author=EJ Anderson&amp;author=NG Rouphael&amp;author=PC Roberts&amp;author=M Makhene&amp;volume=383&amp;publication_year=2020a&amp;pages=1920-1931&amp;pmid=32663912&amp;doi=10.1056/NEJMoa2022483&amp;"/></mixed-citation></ref><ref id="GR278870LIC17"><mixed-citation><named-content content-type="citation-string">Jackson NA, Kester KE, Casimiro D, Gurunathan S, DeRosa F. 2020b. The promise of mRNA vaccines: a biotech and industrial perspective. NPJ Vaccines
5: 11. 10.1038/s41541-020-0159-8
</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1038/s41541-020-0159-8"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC7000814"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="32047656"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=NPJ Vaccines&amp;title=The promise of mRNA vaccines: a biotech and industrial perspective&amp;author=NA Jackson&amp;author=KE Kester&amp;author=D Casimiro&amp;author=S Gurunathan&amp;author=F DeRosa&amp;volume=5&amp;publication_year=2020b&amp;pages=11&amp;pmid=32047656&amp;doi=10.1038/s41541-020-0159-8&amp;"/></mixed-citation></ref><ref id="GR278870LIC18"><mixed-citation><named-content content-type="citation-string">Ji Y, Zhou Z, Liu H, Davuluri RV. 2021. DNABERT: pre-trained bidirectional encoder representations from transformers model for DNA-language in genome. Bioinformatics
37: 2112–2120. 10.1093/bioinformatics/btab083
</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1093/bioinformatics/btab083"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC11025658"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="33538820"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Bioinformatics&amp;title=DNABERT: pre-trained bidirectional encoder representations from transformers model for DNA-language in genome&amp;author=Y Ji&amp;author=Z Zhou&amp;author=H Liu&amp;author=RV Davuluri&amp;volume=37&amp;publication_year=2021&amp;pages=2112-2120&amp;pmid=33538820&amp;doi=10.1093/bioinformatics/btab083&amp;"/></mixed-citation></ref><ref id="GR278870LIC19"><mixed-citation><named-content content-type="citation-string">Kim Y.
2014. Convolutional neural networks for sentence classification. In Proceedings of the 2014 Conference on Empirical Methods in Natural Language Processing (EMNLP) (ed. Moschitti A, et al. ), pp. 1746–1751. Association for Computational Linguistics, Doha, Qatar. <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://aclanthology.org/D14-1181" ext-link-type="uri">https://aclanthology.org/D14-1181</ext-link>. 10.3115/v1/D14-1181</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.3115/v1/D14-1181"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="title=Proceedings of the 2014 Conference on Empirical Methods in Natural Language Processing (EMNLP)&amp;author=Y. Kim&amp;author=A Moschitti&amp;publication_year=2014&amp;"/></mixed-citation></ref><ref id="GR278870LIC20"><mixed-citation><named-content content-type="citation-string">Leppek K, Byeon GW, Kladwang W, Wayment-Steele HK, Kerr CH, Xu AF, Kim DS, Topkar VV, Choe C, Rothschild D, et al. 
2022. Combinatorial optimization of mRNA structure, stability, and translation for RNA-based therapeutics. Nat Commun
13: 1536. 10.1038/s41467-022-28776-w
</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1038/s41467-022-28776-w"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC8940940"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="35318324"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nat Commun&amp;title=Combinatorial optimization of mRNA structure, stability, and translation for RNA-based therapeutics&amp;author=K Leppek&amp;author=GW Byeon&amp;author=W Kladwang&amp;author=HK Wayment-Steele&amp;author=CH Kerr&amp;volume=13&amp;publication_year=2022&amp;pages=1536&amp;pmid=35318324&amp;doi=10.1038/s41467-022-28776-w&amp;"/></mixed-citation></ref><ref id="GR278870LIC21"><mixed-citation><named-content content-type="citation-string">Li S, Zhang H, Zhang L, Liu K, Liu B, Mathews DH, Huang L. 2021. LinearTurboFold: linear-time global prediction of conserved structures for RNA homologs with applications to SARS-CoV-2. Proc Natl Acad Sci
118: e2116269118. 10.1073/pnas.2116269118
</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1073/pnas.2116269118"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC8719904"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="34887342"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Proc Natl Acad Sci&amp;title=LinearTurboFold: linear-time global prediction of conserved structures for RNA homologs with applications to SARS-CoV-2&amp;author=S Li&amp;author=H Zhang&amp;author=L Zhang&amp;author=K Liu&amp;author=B Liu&amp;volume=118&amp;publication_year=2021&amp;pages=e2116269118&amp;pmid=34887342&amp;doi=10.1073/pnas.2116269118&amp;"/></mixed-citation></ref><ref id="GR278870LIC22"><mixed-citation><named-content content-type="citation-string">Lorentzen CL, Haanen JB, Met Ö, Svane IM. 2022. Clinical advances and ongoing trials of mRNA vaccines for cancer treatment. Lancet Oncol
23: e450–e458. 10.1016/S1470-2045(22)00372-2
</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1016/S1470-2045(22)00372-2"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC9512276"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="36174631"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Lancet Oncol&amp;title=Clinical advances and ongoing trials of mRNA vaccines for cancer treatment&amp;author=CL Lorentzen&amp;author=JB Haanen&amp;author=Ö Met&amp;author=IM Svane&amp;volume=23&amp;publication_year=2022&amp;pages=e450-e458&amp;pmid=36174631&amp;doi=10.1016/S1470-2045(22)00372-2&amp;"/></mixed-citation></ref><ref id="GR278870LIC23"><mixed-citation><named-content content-type="citation-string">Maruggi G, Chiarot E, Giovani C, Buccato S, Bonacci S, Frigimelica E, Margarit I, Geall A, Bensi G, Maione D. 2017. Immunogenicity and protective efficacy induced by self-amplifying mRNA vaccines encoding bacterial antigens. Vaccine
35: 361–368. 10.1016/j.vaccine.2016.11.040
</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1016/j.vaccine.2016.11.040"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="27939014"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Vaccine&amp;title=Immunogenicity and protective efficacy induced by self-amplifying mRNA vaccines encoding bacterial antigens&amp;author=G Maruggi&amp;author=E Chiarot&amp;author=C Giovani&amp;author=S Buccato&amp;author=S Bonacci&amp;volume=35&amp;publication_year=2017&amp;pages=361-368&amp;pmid=27939014&amp;doi=10.1016/j.vaccine.2016.11.040&amp;"/></mixed-citation></ref><ref id="GR278870LIC24"><mixed-citation><named-content content-type="citation-string">Mauger DM, Cabral BJ, Presnyak V, Su SV, Reid DW, Goodman B, Link K, Khatwani N, Reynders J, Moore MJ, et al. 
2019. mRNA structure regulates protein expression through changes in functional half-life. Proc Natl Acad Sci
116: 24075–24083. 10.1073/pnas.1908052116
</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1073/pnas.1908052116"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC6883848"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="31712433"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Proc Natl Acad Sci&amp;title=mRNA structure regulates protein expression through changes in functional half-life&amp;author=DM Mauger&amp;author=BJ Cabral&amp;author=V Presnyak&amp;author=SV Su&amp;author=DW Reid&amp;volume=116&amp;publication_year=2019&amp;pages=24075-24083&amp;pmid=31712433&amp;doi=10.1073/pnas.1908052116&amp;"/></mixed-citation></ref><ref id="GR278870LIC25"><mixed-citation><named-content content-type="citation-string">Mauro VP. 2018. Codon optimization in the production of recombinant biotherapeutics: potential risks and considerations. BioDrugs
32: 69–81. 10.1007/s40259-018-0261-x
</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1007/s40259-018-0261-x"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="29392566"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=BioDrugs&amp;title=Codon optimization in the production of recombinant biotherapeutics: potential risks and considerations&amp;author=VP Mauro&amp;volume=32&amp;publication_year=2018&amp;pages=69-81&amp;pmid=29392566&amp;doi=10.1007/s40259-018-0261-x&amp;"/></mixed-citation></ref><ref id="GR278870LIC26"><mixed-citation><named-content content-type="citation-string">Mauro VP, Chappell SA. 2014. A critical analysis of codon optimization in human therapeutics. Trends Mol Med
20: 604–613. 10.1016/j.molmed.2014.09.003
</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1016/j.molmed.2014.09.003"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC4253638"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="25263172"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Trends Mol Med&amp;title=A critical analysis of codon optimization in human therapeutics&amp;author=VP Mauro&amp;author=SA Chappell&amp;volume=20&amp;publication_year=2014&amp;pages=604-613&amp;pmid=25263172&amp;doi=10.1016/j.molmed.2014.09.003&amp;"/></mixed-citation></ref><ref id="GR278870LIC27"><mixed-citation><named-content content-type="citation-string">McInnes L, Healy J, Saul N, Grossberger L. 2018. UMAP: Uniform Manifold Approximation and Projection. J Open Source Softw
3: 861. 10.21105/joss.00861</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.21105/joss.00861"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=J Open Source Softw&amp;title=UMAP: Uniform Manifold Approximation and Projection&amp;author=L McInnes&amp;author=J Healy&amp;author=N Saul&amp;author=L Grossberger&amp;volume=3&amp;publication_year=2018&amp;pages=861&amp;doi=10.21105/joss.00861&amp;"/></mixed-citation></ref><ref id="GR278870LIC28"><mixed-citation><named-content content-type="citation-string">Miao L, Zhang Y, Huang L. 2021. mRNA vaccine for cancer immunotherapy. Mol Cancer
20: 41. 10.1186/s12943-021-01335-5
</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1186/s12943-021-01335-5"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC7905014"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="33632261"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Mol Cancer&amp;title=mRNA vaccine for cancer immunotherapy&amp;author=L Miao&amp;author=Y Zhang&amp;author=L Huang&amp;volume=20&amp;publication_year=2021&amp;pages=41&amp;pmid=33632261&amp;doi=10.1186/s12943-021-01335-5&amp;"/></mixed-citation></ref><ref id="GR278870LIC29"><mixed-citation><named-content content-type="citation-string">Mikolov T, Chen K, Corrado G, Dean J. 2013. Efficient estimation of word representations in vector space. arXiv: 1301.3781 [cs.CL].</named-content></mixed-citation></ref><ref id="GR278870LIC30"><mixed-citation><named-content content-type="citation-string">Nieuwkoop T, Terlouw BR, Stevens KG, Scheltema R, de Ridder D, van der Oost J, Claassens N. 2023. Revealing determinants of translation efficiency via whole-gene codon randomization and machine learning. Nucleic Acids Res
51: 2363–2376. 10.1093/nar/gkad035
</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1093/nar/gkad035"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC10018363"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="36718935"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nucleic Acids Res&amp;title=Revealing determinants of translation efficiency via whole-gene codon randomization and machine learning&amp;author=T Nieuwkoop&amp;author=BR Terlouw&amp;author=KG Stevens&amp;author=R Scheltema&amp;author=D de Ridder&amp;volume=51&amp;publication_year=2023&amp;pages=2363-2376&amp;pmid=36718935&amp;doi=10.1093/nar/gkad035&amp;"/></mixed-citation></ref><ref id="GR278870LIC31"><mixed-citation><named-content content-type="citation-string">Pardi N, Hogan MJ, Pelc RS, Muramatsu H, Andersen H, DeMaso CR, Dowd KA, Sutherland LL, Scearce RM, Parks R, et al. 
2017. Zika virus protection by a single low-dose nucleoside-modified mRNA vaccination. Nature
543: 248–251. 10.1038/nature21428
</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1038/nature21428"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC5344708"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="28151488"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nature&amp;title=Zika virus protection by a single low-dose nucleoside-modified mRNA vaccination&amp;author=N Pardi&amp;author=MJ Hogan&amp;author=RS Pelc&amp;author=H Muramatsu&amp;author=H Andersen&amp;volume=543&amp;publication_year=2017&amp;pages=248-251&amp;pmid=28151488&amp;doi=10.1038/nature21428&amp;"/></mixed-citation></ref><ref id="GR278870LIC32"><mixed-citation><named-content content-type="citation-string">Pardi N, Hogan MJ, Porter FW, Weissman D. 2018. mRNA vaccines: a new era in vaccinology. Nat Rev Drug Discov
17: 261–279. 10.1038/nrd.2017.243
</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1038/nrd.2017.243"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC5906799"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="29326426"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nat Rev Drug Discov&amp;title=mRNA vaccines: a new era in vaccinology&amp;author=N Pardi&amp;author=MJ Hogan&amp;author=FW Porter&amp;author=D Weissman&amp;volume=17&amp;publication_year=2018&amp;pages=261-279&amp;pmid=29326426&amp;doi=10.1038/nrd.2017.243&amp;"/></mixed-citation></ref><ref id="GR278870LIC33"><mixed-citation><named-content content-type="citation-string">Pardi N, Hogan MJ, Weissman D. 2020. Recent advances in mRNA vaccine technology. Curr Opin Immunol
65: 14–20. 10.1016/j.coi.2020.01.008
</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1016/j.coi.2020.01.008"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="32244193"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Curr Opin Immunol&amp;title=Recent advances in mRNA vaccine technology&amp;author=N Pardi&amp;author=MJ Hogan&amp;author=D Weissman&amp;volume=65&amp;publication_year=2020&amp;pages=14-20&amp;pmid=32244193&amp;doi=10.1016/j.coi.2020.01.008&amp;"/></mixed-citation></ref><ref id="GR278870LIC34"><mixed-citation><named-content content-type="citation-string">Parret AH, Besir H, Meijers R. 2016. Critical reflections on synthetic gene design for recombinant protein expression. Curr Opin Struct Biol
38: 155–162. 10.1016/j.sbi.2016.07.004
</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1016/j.sbi.2016.07.004"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="27449695"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Curr Opin Struct Biol&amp;title=Critical reflections on synthetic gene design for recombinant protein expression&amp;author=AH Parret&amp;author=H Besir&amp;author=R Meijers&amp;volume=38&amp;publication_year=2016&amp;pages=155-162&amp;pmid=27449695&amp;doi=10.1016/j.sbi.2016.07.004&amp;"/></mixed-citation></ref><ref id="GR278870LIC35"><mixed-citation><named-content content-type="citation-string">Peters ME, Neumann M, Iyyer M, Gardner M, Clark C, Lee K, Zettlemoyer L. 2018. Deep contextualized word representations. In Proceedings of the 2018 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long Papers) (ed. Walker M, et al.), pp. 2227–2237. Association for Computational Linguistics, New Orleans. <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://aclanthology.org/N18-1202" ext-link-type="uri">https://aclanthology.org/N18-1202</ext-link>. 10.18653/v1/N18-1202</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.18653/v1/N18-1202"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="title=Proceedings of the 2018 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long Papers) (ed. Walker M, et al.)&amp;author=ME Peters&amp;author=M Neumann&amp;author=M Iyyer&amp;author=M Gardner&amp;author=C Clark&amp;publication_year=2018&amp;"/></mixed-citation></ref><ref id="GR278870LIC36"><mixed-citation><named-content content-type="citation-string">Pilkington EH, Suys EJ, Trevaskis NL, Wheatley AK, Zukancic D, Algarni A, Al-Wassiti H, Davis TP, Pouton CW, Kent SJ, et al. 
2021. From influenza to COVID-19: lipid nanoparticle mRNA vaccines at the frontiers of infectious diseases. Acta Biomater
131: 16–40. 10.1016/j.actbio.2021.06.023
</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1016/j.actbio.2021.06.023"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC8272596"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="34153512"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Acta Biomater&amp;title=From influenza to COVID-19: lipid nanoparticle mRNA vaccines at the frontiers of infectious diseases&amp;author=EH Pilkington&amp;author=EJ Suys&amp;author=NL Trevaskis&amp;author=AK Wheatley&amp;author=D Zukancic&amp;volume=131&amp;publication_year=2021&amp;pages=16-40&amp;pmid=34153512&amp;doi=10.1016/j.actbio.2021.06.023&amp;"/></mixed-citation></ref><ref id="GR278870LIC37"><mixed-citation><named-content content-type="citation-string">Radford A, Narasimhan K, Salimans T, Sutskever I. 2018. Improving language understanding by generative pre-training. OpenAI. <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://cdn.openai.com/research-covers/language-unsupervised/language_understanding_paper.pdf" ext-link-type="uri">https://cdn.openai.com/research-covers/language-unsupervised/language_understanding_paper.pdf</ext-link>.</named-content></mixed-citation></ref><ref id="GR278870LIC38"><mixed-citation><named-content content-type="citation-string">Rajaraman A, Ullman JD. 2011. Mining of massive datasets. Cambridge University Press, Cambridge.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="title=Mining of massive datasets&amp;author=A Rajaraman&amp;author=JD Ullman&amp;publication_year=2011&amp;"/></mixed-citation></ref><ref id="GR278870LIC39"><mixed-citation><named-content content-type="citation-string">Řehůřek R, Sojka P.
2010. Software framework for topic modelling with large corpora. In Proceedings of the LREC 2010 Workshop on New Challenges for NLP Frameworks, pp. 45–50. ELRA, Valletta, Malta. <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="http://is.muni.cz/publication/884893/en" ext-link-type="uri">http://is.muni.cz/publication/884893/en</ext-link>.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="title=Proceedings of the LREC 2010 Workshop on New Challenges for NLP Frameworks&amp;author=R Řehůřek&amp;author=P. Sojka&amp;publication_year=2010&amp;"/></mixed-citation></ref><ref id="GR278870LIC40"><mixed-citation><named-content content-type="citation-string">Rives A, Meier J, Sercu T, Goyal S, Lin Z, Liu J, Guo D, Ott M, Zitnick CL, Ma J, et al. 
2021. Biological structure and function emerge from scaling unsupervised learning to 250 million protein sequences. Proc Natl Acad Sci
118: e2016239118. 10.1073/pnas.2016239118
</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1073/pnas.2016239118"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC8053943"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="33876751"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Proc Natl Acad Sci&amp;title=Biological structure and function emerge from scaling unsupervised learning to 250 million protein sequences&amp;author=A Rives&amp;author=J Meier&amp;author=T Sercu&amp;author=S Goyal&amp;author=Z Lin&amp;volume=118&amp;publication_year=2021&amp;pages=e2016239118&amp;pmid=33876751&amp;doi=10.1073/pnas.2016239118&amp;"/></mixed-citation></ref><ref id="GR278870LIC41"><mixed-citation><named-content content-type="citation-string">Schlake T, Thess A, Fotin-Mleczek M, Kallen K-J. 2012. Developing mRNA-vaccine technologies. RNA Biol
9: 1319–1330. 10.4161/rna.22269
</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.4161/rna.22269"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC3597572"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="23064118"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=RNA Biol&amp;title=Developing mRNA-vaccine technologies&amp;author=T Schlake&amp;author=A Thess&amp;author=M Fotin-Mleczek&amp;author=K-J Kallen&amp;volume=9&amp;publication_year=2012&amp;pages=1319-1330&amp;pmid=23064118&amp;doi=10.4161/rna.22269&amp;"/></mixed-citation></ref><ref id="GR278870LIC42"><mixed-citation><named-content content-type="citation-string">Schmidt M, Hamacher K, Reinhardt F, Lotz TS, Groher F, Suess B, Jager S. 2020. SICOR: subgraph isomorphism comparison of RNA secondary structures. IEEE/ACM Trans Comput Biol Bioinform
17: 2189–2195. 10.1109/TCBB.2019.2926711
</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1109/TCBB.2019.2926711"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="31295116"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=IEEE/ACM Trans Comput Biol Bioinform&amp;title=SICOR: subgraph isomorphism comparison of RNA secondary structures&amp;author=M Schmidt&amp;author=K Hamacher&amp;author=F Reinhardt&amp;author=TS Lotz&amp;author=F Groher&amp;volume=17&amp;publication_year=2020&amp;pages=2189-2195&amp;pmid=31295116&amp;doi=10.1109/TCBB.2019.2926711&amp;"/></mixed-citation></ref><ref id="GR278870LIC44"><mixed-citation><named-content content-type="citation-string">Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser Ł, Polosukhin I. 2017. Attention is all you need. In <italic>31st Conference on Neutral Information Processing Systems (NIPS 2017)</italic>, Long Beach, CA, pp. 6000–6010.</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="Vaswani A, Shazeer N, Parmar N, Uszkoreit J, Jones L, Gomez AN, Kaiser Ł, Polosukhin I. 2017. Attention is all you need. In 31st Conference on Neutral Information Processing Systems (NIPS 2017), Long Beach, CA, pp. 6000–6010."/></mixed-citation></ref><ref id="GR278870LIC45"><mixed-citation><named-content content-type="citation-string">Wayment-Steele HK, Kladwang W, Watkins AM, Kim DS, Tunguz B, Reade W, Demkin M, Romano J, Wellington-Oguri R, Nicol JJ, et al. 
2022. Deep learning models for predicting RNA degradation via dual crowdsourcing. Nat Mach Intell
4: 1174–1184. 10.1038/s42256-022-00571-8
</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1038/s42256-022-00571-8"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC9771809"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="36567960"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nat Mach Intell&amp;title=Deep learning models for predicting RNA degradation via dual crowdsourcing&amp;author=HK Wayment-Steele&amp;author=W Kladwang&amp;author=AM Watkins&amp;author=DS Kim&amp;author=B Tunguz&amp;volume=4&amp;publication_year=2022&amp;pages=1174-1184&amp;pmid=36567960&amp;doi=10.1038/s42256-022-00571-8&amp;"/></mixed-citation></ref><ref id="GR278870LIC46"><mixed-citation><named-content content-type="citation-string">Webster GR, Teh AY-H, Ma JK-C. 2017. Synthetic gene design: the rationale for codon optimization and implications for molecular pharming in plants. Biotechnol Bioeng
114: 492–502. 10.1002/bit.26183
</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1002/bit.26183"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="27618314"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Biotechnol Bioeng&amp;title=Synthetic gene design: the rationale for codon optimization and implications for molecular pharming in plants&amp;author=GR Webster&amp;author=AY-H Teh&amp;author=JK-C Ma&amp;volume=114&amp;publication_year=2017&amp;pages=492-502&amp;pmid=27618314&amp;doi=10.1002/bit.26183&amp;"/></mixed-citation></ref><ref id="GR278870LIC47"><mixed-citation><named-content content-type="citation-string">Wheeler DL, Barrett T, Benson DA, Bryant SH, Canese K, Chetvernin V, Church DM, DiCuccio M, Edgar R, Federhen S, et al. 
2007. Database resources of the National Center for Biotechnology Information. Nucleic Acids Res
35: D5–D12. 10.1093/nar/gkl1031
</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1093/nar/gkl1031"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC1781113"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="17170002"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nucleic Acids Res&amp;title=Database resources of the National Center for Biotechnology Information&amp;author=DL Wheeler&amp;author=T Barrett&amp;author=DA Benson&amp;author=SH Bryant&amp;author=K Canese&amp;volume=35&amp;publication_year=2007&amp;pages=D5-D12&amp;pmid=17170002&amp;doi=10.1093/nar/gkl1031&amp;"/></mixed-citation></ref><ref id="GR278870LIC48"><mixed-citation><named-content content-type="citation-string">Wint R, Salamov A, Grigoriev IV. 2022. Kingdom-wide analysis of fungal protein-coding and tRNA genes reveals conserved patterns of adaptive evolution. Mol Biol Evol
39: msab372. 10.1093/molbev/msab372
</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1093/molbev/msab372"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC8826637"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="35060603"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Mol Biol Evol&amp;title=Kingdom-wide analysis of fungal protein-coding and tRNA genes reveals conserved patterns of adaptive evolution&amp;author=R Wint&amp;author=A Salamov&amp;author=IV Grigoriev&amp;volume=39&amp;publication_year=2022&amp;pages=msab372&amp;pmid=35060603&amp;doi=10.1093/molbev/msab372&amp;"/></mixed-citation></ref><ref id="GR278870LIC49"><mixed-citation><named-content content-type="citation-string">Zhang C, Maruggi G, Shan H, Li J. 2019. Advances in mRNA vaccines for infectious diseases. Front Immunol
10: 594. 10.3389/fimmu.2019.00594
</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.3389/fimmu.2019.00594"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC6446947"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="30972078"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Front Immunol&amp;title=Advances in mRNA vaccines for infectious diseases&amp;author=C Zhang&amp;author=G Maruggi&amp;author=H Shan&amp;author=J Li&amp;volume=10&amp;publication_year=2019&amp;pages=594&amp;pmid=30972078&amp;doi=10.3389/fimmu.2019.00594&amp;"/></mixed-citation></ref><ref id="GR278870LIC50"><mixed-citation><named-content content-type="citation-string">Zhang H, Zhang L, Lin A, Xu C, Li Z, Liu K, Liu B, Ma X, Zhao F, Jiang H, et al. 
2023. Algorithm for optimized mRNA design improves stability and immunogenicity. Nature
621: 396–403. 10.1038/s41586-023-06127-z
</named-content><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="doi" xlink:href="10.1038/s41586-023-06127-z"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmcid" xlink:href="PMC10499610"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="pmid" xlink:href="37130545"/><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="google-scholar" xlink:href="journal=Nature&amp;title=Algorithm for optimized mRNA design improves stability and immunogenicity&amp;author=H Zhang&amp;author=L Zhang&amp;author=A Lin&amp;author=C Xu&amp;author=Z Li&amp;volume=621&amp;publication_year=2023&amp;pages=396-403&amp;pmid=37130545&amp;doi=10.1038/s41586-023-06127-z&amp;"/></mixed-citation></ref></ref-list></sec></sec><sec id="_ad93_" xml:lang="en" sec-type="associated-data" disp-level="1"><title>Associated Data</title><sec id="_adsm93_" xml:lang="en" sec-type="supplementary-materials" disp-level="2"><title>Supplementary Materials</title><supplementary-material id="db_ds_supplementary-material1_reqid_" position="float"><?disp-level 2?><label>Supplement 1</label><media xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="Supplemental_Material.zip" mimetype="application" mime-subtype="zip"><?cloudpmc-path d27d/11368176/365fed031c19/Supplemental_Material.zip?><?cloudpmc-bucket app?><?size 4754142?></media></supplementary-material><supplementary-material id="db_ds_supplementary-material2_reqid_" position="float"><?disp-level 2?><label>Supplement 2</label><media xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="Supplemental_Source_Code.zip" mimetype="application" mime-subtype="zip"><?cloudpmc-path d27d/11368176/6a97e4d854b0/Supplemental_Source_Code.zip?><?cloudpmc-bucket app?><?size 192998894?></media></supplementary-material></sec></sec></body></article>