
<!DOCTYPE article
  PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD with MathML3 v1.4 20241031//EN" "JATS-archivearticle1-4-mathml3.dtd">
<article xml:lang="en" article-type="research-article" dtd-version="1.4"><processing-meta base-tagset="archiving" mathml-version="3.0" table-model="xhtml" tagset-family="jats"><restricted-by>pmc</restricted-by></processing-meta><front><journal-meta><journal-id journal-id-type="nlm-ta">Brief Bioinform</journal-id><journal-id journal-id-type="iso-abbrev">Brief Bioinform</journal-id><journal-id journal-id-type="pmc-domain-id">721</journal-id><journal-id journal-id-type="pmc-domain">bib</journal-id><journal-id journal-id-type="nlm-id">100912837</journal-id><journal-id journal-id-type="publisher-id">bib</journal-id><journal-title-group><journal-title>Briefings in Bioinformatics</journal-title></journal-title-group><issn pub-type="ppub">1467-5463</issn><issn pub-type="epub">1477-4054</issn><?publisher_abbrev oup?><publisher><publisher-name>Oxford University Press</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="pmcid">PMC12790621</article-id><article-id pub-id-type="pmcid-ver">PMC12790621.1</article-id><article-id pub-id-type="pmcaid">12790621</article-id><article-id pub-id-type="pmcaiid">12790621</article-id><article-id pub-id-type="pmid">41520231</article-id><article-id pub-id-type="doi">10.1093/bib/bbaf713</article-id><article-id pub-id-type="publisher-id">bbaf713</article-id><article-version article-version-type="pmc-version">1</article-version><article-categories><subj-group subj-group-type="heading"><subject>Problem Solving Protocol</subject></subj-group><subj-group subj-group-type="category-taxonomy-collection"><subject>AcademicSubjects/SCI01060</subject></subj-group></article-categories><title-group><article-title>CoBRA: compound binding site prediction using RNA language model</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Jang</surname><given-names initials="W">Wonkyeong</given-names></name><aff>
<institution>Department of Biomedical Informatics, Korea University College of Medicine</institution>, <addr-line>161 Jeongneung-ro, Seongbuk-gu, Seoul 02708</addr-line>, <country country="KR">Republic of Korea</country></aff></contrib><contrib contrib-type="author" corresp="yes"><contrib-id contrib-id-type="orcid" authenticated="false">https://orcid.org/0000-0003-3462-0243</contrib-id><name name-style="western"><surname>Shin</surname><given-names initials="WH">Woong-Hee</given-names></name><aff>
<institution>Department of Biomedical Informatics, Korea University College of Medicine</institution>, <addr-line>161 Jeongneung-ro, Seongbuk-gu, Seoul 02708</addr-line>, <country country="KR">Republic of Korea</country></aff><aff>
<institution>Arontier, Co.</institution>, <addr-line>241 Gangnam-daero, Seocho-gu, Seoul 06735</addr-line>, <country country="KR">Republic of Korea</country></aff><xref rid="cor1" ref-type="corresp"/></contrib></contrib-group><author-notes><corresp id="cor1">Corresponding author. Department of Biomedical Informatics, Korea University College of Medicine, 161 Jeongneung-ro, Seongbuk-gu, Seoul 02708, Republic of Korea. E-mail: <email>whshin@korea.ac.kr</email></corresp></author-notes><pub-date pub-type="collection"><month>1</month><year>2026</year></pub-date><pub-date pub-type="epub" iso-8601-date="2026-01-11"><day>11</day><month>1</month><year>2026</year></pub-date><volume>27</volume><issue>1</issue><issue-id pub-id-type="pmc-issue-id">504440</issue-id><elocation-id>bbaf713</elocation-id><history><date date-type="received"><day>22</day><month>9</month><year>2025</year></date><date date-type="rev-recd"><day>22</day><month>11</month><year>2025</year></date><date date-type="accepted"><day>16</day><month>12</month><year>2025</year></date></history><pub-history><event event-type="pmc-release"><date><day>11</day><month>01</month><year>2026</year></date></event><event event-type="pmc-live"><date><day>12</day><month>01</month><year>2026</year></date></event><event event-type="pmc-last-change"><date iso-8601-date="2026-01-12 09:25:14.390"><day>12</day><month>01</month><year>2026</year></date></event></pub-history><permissions><copyright-statement>© The Author(s) 2026. Published by Oxford University Press.</copyright-statement><copyright-year>2026</copyright-year><license><ali:license_ref xmlns:ali="http://www.niso.org/schemas/ali/1.0/" specific-use="textmining" content-type="ccbynclicense">https://creativecommons.org/licenses/by-nc/4.0/</ali:license_ref><license-p>This is an Open Access article distributed under the terms of the Creative Commons Attribution-NonCommercial License (<ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by-nc/4.0/">https://creativecommons.org/licenses/by-nc/4.0/</ext-link>), which permits non-commercial re-use, distribution, and reproduction in any medium, provided the original work is properly cited. For commercial re-use, please contact reprints@oup.com for reprints and translation rights for reprints. All other permissions can be obtained through our RightsLink service via the Permissions link on the article page on our site—for further information please contact journals.permissions@oup.com.</license-p></license></permissions><self-uri xmlns:xlink="http://www.w3.org/1999/xlink" content-type="pmc-pdf" xlink:href="bbaf713.pdf"><?pdf-name bbaf713.pdf?><?pdf-size 828834?><?pdf-md5 3512cc69fd0ab5bd085983d556d2cae5?><?pdf-image-server-status NEVER_LOAD?><?pdf-cloudpmc-urn urn:app:9241/12790621/3512cc69fd0a/bbaf713.pdf?></self-uri><self-uri xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="bbaf713.pdf"/><abstract><title>Abstract</title><p>RNA performs a variety of functions within cells and is implicated in various human diseases. Because druggable proteins occupy a small portion of the genome, considerable interest has been increasing in developing drugs targeting RNAs. Thus, precise prediction of small-molecule binding sites across different classes of RNAs is important. In this study, a lightweight deep learning program for predicting RNA-drug binding sites, called compound binding site prediction for RNA (CoBRA), is introduced. Our approach utilizes residue-level embeddings derived from a pre-trained RNA language model, without relying on any structural information. These embeddings encapsulate the contextual and statistical properties of each nucleotide and are used as input for a multi-layer perceptron classifier that performs binary classification of binding nucleotides. The model was trained using the TR60 and HARIBOSS datasets and tested on four independent benchmark sets. The performance of CoBRA demonstrates a relative improvement of 22.1% in the Matthew correlation coefficient and a 45.6% increase in sensitivity compared to existing state-of-the-art RNA–ligand binding site prediction methods that utilize structural information. These results demonstrate that sequence-based language model embeddings, which do not require explicit coordinate or distance information, can match or outperform structure-based methods. This makes it a flexible tool for predicting binding sites across diverse RNA targets.</p></abstract><abstract abstract-type="graphical"><title>Graphical Abstract</title><p>
<fig position="float" id="ga1" orientation="portrait"><label>Graphical Abstract</label><graphic xmlns:xlink="http://www.w3.org/1999/xlink" position="float" orientation="portrait" xlink:href="bbaf713ga1.jpg"><?image-name bbaf713ga1.jpg?><?image-size 55664?><?image-md5 0ff1812db461cce5c48ec605e10b23b5?><?image-image-server-status LOAD_COMPLETED?><?image-original-height 1293?><?image-original-width 3838?><?image-scaled-height 258?><?image-scaled-width 767?><?image-cloudpmc-urn urn:cdn:blobs/9241/12790621/0ff1812db461/bbaf713ga1.jpg?><?thumb-name bbaf713ga1.gif?><?thumb-size 6009?><?thumb-md5 d2ed3de53704ff77620350d3927efc8b?><?thumb-image-server-status NEVER_LOAD?><?thumb-scaled-height 67?><?thumb-scaled-width 200?><?thumb-cloudpmc-urn urn:cdn:blobs/9241/12790621/d2ed3de53704/bbaf713ga1.gif?></graphic></fig>
</p></abstract><kwd-group><kwd>RNA–small molecule binding site prediction</kwd><kwd>RNA language model</kwd><kwd>pre-trained embedding</kwd><kwd>deep learning</kwd><kwd>convolutional neural network</kwd></kwd-group><funding-group><award-group award-type="grant"><funding-source>
<institution-wrap><institution>Institute of Information &amp; Communications Technology Planning &amp; Evaluation (IITP)—ICT Challenge and Advanced Network of HRD (ICAN)</institution></institution-wrap>
</funding-source><award-id>IITP-2025-RS-2022-00156439</award-id><award-id>IITP-2025-RS-2024-00438263</award-id></award-group><award-group award-type="grant"><funding-source>
<institution-wrap><institution>Bio &amp; Medical Technology Development Program of the National Research Foundation (NRF)</institution></institution-wrap>
</funding-source><award-id>RS-2025-02217289</award-id><award-id>2022M3E5F3081268</award-id></award-group><award-group award-type="grant"><funding-source>
<institution-wrap><institution>Korea University</institution><institution-id institution-id-type="DOI">10.13039/501100002642</institution-id></institution-wrap>
</funding-source><award-id>K2517281</award-id></award-group></funding-group><counts><page-count count="10"/></counts><custom-meta-group><custom-meta><meta-name>pmc-status-qastatus</meta-name><meta-value>0</meta-value></custom-meta><custom-meta><meta-name>pmc-status-live</meta-name><meta-value>yes</meta-value></custom-meta><custom-meta><meta-name>pmc-status-embargo</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-status-released</meta-name><meta-value>yes</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-open-access</meta-name><meta-value>yes</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-olf</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-manuscript</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-legally-suppressed</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-has-pdf</meta-name><meta-value>yes</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-has-supplement</meta-name><meta-value>yes</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-pdf-only</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-suppress-copyright</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-is-real-version</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-is-scanned-article</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-preprint</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-in-epmc</meta-name><meta-value>yes</meta-value></custom-meta><custom-meta><meta-name>pmc-license-ref</meta-name><meta-value>CC BY-NC</meta-value></custom-meta></custom-meta-group></article-meta></front><body><sec id="sec4"><title>Introduction</title><p>RNA plays diverse roles in the cell’s gene regulation, translation, and structural organization. They are also involved in various human diseases, including cancer, neurological disorders, cardiovascular dysfunction, and developmental abnormalities [<xref rid="ref1" ref-type="bibr">1–3</xref>]. Given that only 1%–3% of the human genome encodes proteins [<xref rid="ref4" ref-type="bibr">4</xref>] and that ~10%–14% of these proteins are considered druggable, leaving the majority (80%–90%) undruggable [<xref rid="ref5" ref-type="bibr">5</xref>], targeting mRNAs has recently gained attention as a potential therapeutic strategy. Additionally, there has been an escalating focus on their potential as small-molecule drug targets, owing to their capacity to generate structured regions for binding with high specificity [<xref rid="ref6" ref-type="bibr">6</xref>, <xref rid="ref7" ref-type="bibr">7</xref>]. These RNA-ligand interactions present novel avenues for therapeutic intervention, and thus, a precise computational prediction of ligand-binding sites across diverse RNA categories is needed.</p><p>Various RNA-ligand binding site prediction methods relying on structural information, either directly or indirectly, have been developed. Rsite [<xref rid="ref8" ref-type="bibr">8</xref>, <xref rid="ref9" ref-type="bibr">9</xref>] calculates Euclidean distances between nucleotides based on RNA tertiary structures and Hamming distances between the secondary structures to predict binding sites. Rbind [<xref rid="ref10" ref-type="bibr">10</xref>] represents RNA tertiary structures as a network, where nucleotides are treated as nodes and non-covalent spatial interactions form edges. Binding site prediction is then performed based on network centrality metrics such as degree and closeness. RNAsite [<xref rid="ref11" ref-type="bibr">11</xref>] is a machine learning-based method that extracts sequence and structure-derived features of nucleotides with a sliding window strategy. A random forest classifier is employed to predict binding status based on these features. ZHmolReSTasite (ZeSTa) [<xref rid="ref12" ref-type="bibr">12</xref>] utilizes point clouds of solvent-excluded surfaces generated from RNA tertiary structures. These are then converted into normalized topographic images corresponding to individual nucleotides. The resulting representations are used as input features to a deep learning model that learns surface-based geometric features. The structure-based approaches have shown solid performance in RNA–ligand binding site prediction by incorporating explicit three-dimensional or secondary structural information into geometric or graph-based representations. However, their reliance on experimentally determined RNA structures restricts their applicability to RNAs without reliable structural data and limits large-scale use.</p><p>On the other side, a couple of prediction methods utilize the RNA language models (LMs). RNABind [<xref rid="ref13" ref-type="bibr">13</xref>] employs E(3)-equivariant graph neural networks to encode both sequence and structure information. Node features are enriched with embeddings from RNA language models, enabling the model to learn both geometric and contextual relationships for identifying ligand-binding nucleotides. It evaluated eight LMs, including ERNIE-RNA and RiNALMo, which achieved superior performance. RLSite [<xref rid="ref14" ref-type="bibr">14</xref>] and GATRsite [<xref rid="ref15" ref-type="bibr">15</xref>] combine RNA LM with graph attention networks. The three-dimensional structure is represented as graphs where nucleotides are nodes, incorporating both sequential features from language models and structural features from three-dimensional conformations. MVRBind [<xref rid="ref16" ref-type="bibr">16</xref>] employs a multi-view feature extraction module, combining sequence information from one-hot, MSA, and LM, secondary structure, and tertiary structure. The gathered information is integrated by using a graph convolutional network. By performing multi-view graph message passing and feature fusion across these representations, the model captures hierarchical geometric dependencies to predict RNA–ligand binding sites. By combining molecular graphs and LMs, these methods achieved high performance.</p><p>In this study, we propose a lightweight deep learning-based model, called CoBRA (Compound Binding Site Prediction for RNA), that predicts RNA–ligand binding sites using an RNA LM, not relying on the RNA structural information. The model utilizes residue-level embeddings derived from pre-trained RNA language models, which implicitly encode the contextual and statistical properties of each nucleotide. These embeddings are subsequently entered into a multi-layer perceptron (MLP) classifier for residue-level binary classification. The proposed framework is designed to operate without explicit structural features such as three-dimensional coordinates or distance metrics, and its modularity allows for flexible experimentation. A combination of ten RNA LMs and six loss functions was systematically trained, combining TR60 [<xref rid="ref11" ref-type="bibr">11</xref>] and HARIBOSS set [<xref rid="ref17" ref-type="bibr">17</xref>], and then evaluated to identify optimal configurations for the RNA-ligand binding site prediction. The final model was evaluated in four benchmark sets: TE18, RB9, JL10, and TL12 [<xref rid="ref11" ref-type="bibr">11</xref>, <xref rid="ref12" ref-type="bibr">12</xref>]. CoBRA achieved a relative improvement compared to the state-of-the-art RNA-ligand binding site prediction programs.</p></sec><sec id="sec5"><title>Material and methods</title><sec id="sec6"><title>Dataset preparation</title><p>The prediction of RNA–ligand binding sites is a binary classification task at the residue level. The model predicts the binding of each nucleotide (residue) in each RNA sequence to an organic small molecule or metal ion. Each residue is designated as binding or non-binding. A nucleotide is labelled as a binding site if the distance between any of its heavy atoms and any ligand heavy atoms is less than or equal to 4 Å, following the definition from previous research [<xref rid="ref12" ref-type="bibr">12</xref>, <xref rid="ref13" ref-type="bibr">13</xref>, <xref rid="ref16" ref-type="bibr">16</xref>].</p><p>A total of six publicly available RNA–ligand complex structure datasets are utilized in this study: HARIBOSS, TR60, RB9, TL12, JL10, and TE18. Panei <italic toggle="yes">et al.</italic> [<xref rid="ref17" ref-type="bibr">17</xref>] extracted RNA-small molecule complexes from the PDB to construct the HARIBOSS dataset (Accessed February 2025), composed of 862 structures. Subsequently, these complexes were clustered based on their sequence and structural similarity with RNA. Su <italic toggle="yes">et al.</italic> [<xref rid="ref11" ref-type="bibr">11</xref>] also collected 712 RNA-small molecule crystal structures and used TM-scoreRNA to calculate their structural similarity, resulting in 78 representatives. The dataset is further segmented into two distinct sets: the TR60 set and the TE18 set. These sets are utilized for the training and testing of RNAsite. Gao <italic toggle="yes">et al.</italic> [<xref rid="ref12" ref-type="bibr">12</xref>] generated RB9, TL12, and JL10 datasets to assess their binding site prediction program, ZeSTa. The RB9 originated from RB19 by removing ten structures from the dataset that overlapped with TR60. From the RNA-small molecule structures deposited after January 2021, JL10 is characterized by junction loop structures, which exhibit high structural complexity. In contrast, TL12 structures are devoid of the loop, resulting in low structural complexity. Among the six datasets, HARIBOSS and TR60 were merged to generate training, test, and validation sets. Non-biological small molecules commonly used as crystallization additives or cryoprotectants, including water, sulfate, phosphate, glycerol, ethylene glycol, and polyethylene glycol (HOH, SO₄, PO₄, GOL, EDO, PEG), were excluded from the ligand set to prevent artifactual contacts. RNA chains longer than 161 nucleotides were subsequently removed, resulting in a final dataset of 432 unique RNA chains. The dataset was segmented using a pre-determined random seed to create three subsets with a ratio of 8:1:1 for training, validation, and internal testing, respectively.</p><p>The HARIBOSS set was utilized to further evaluate the generalizability of the proposed model. Zhu <italic toggle="yes">et al.</italic> [<xref rid="ref13" ref-type="bibr">13</xref>] provide the four structure split sets based on structural similarity with a TM-Score threshold of 0.5. Sixty models combining RNA LMs and loss functions were trained and tested using the same protocol as the combined sets of HARIBOSS and TR60.</p><p>The remaining four, RB9, JL10, TL12, and TE18, were reserved as test sets. To ensure independent evaluation, any overlapping sequences between the training and test sets were removed. The problem formulation and dataset design enable quantitative assessment of both prediction accuracy and generalizability under practically meaningful scenarios.</p></sec><sec id="sec7"><title>Model architecture</title><p>To predict whether each nucleotide in an RNA sequence is a binding site or not, a residue-level binary classification model was designed. <xref rid="f1" ref-type="fig">Fig. 1</xref> illustrates the model architecture of CoBRA. The program’s input is a query RNA sequence, which is embedded using pre-trained RNA LMs. During the training process, these embeddings are maintained as a frozen state and do not undergo updates. The prediction model employs an MLP architecture comprising five fully connected layers. Each layer is followed sequentially by layer normalization, a ReLU activation function, and dropout with a probability of .1, promoting stable training and improved generalization. The final output layer produces a two-dimensional logit for each residue, corresponding to the binding and non-binding classes. These are converted into probabilities via a SoftMax function. All inputs are standardized to a maximum sequence length of 161 residues. Sequences of shorter length are padded with zeros in the input matrix X and with −1 in the target vector y. A masking strategy is implemented to ensure that the padding regions do not contribute to the loss computation or affect model training.</p><fig position="float" id="f1" orientation="portrait"><label>Figure 1</label><caption><p>An overview of the model architecture. The model takes a sequence of 161 input embeddings, each zero-padded to a fixed embedding dimension. The sequence is processed by an MLP with hidden sizes of 1024, 256, 128, and 64. Each layer includes layer normalization, ReLU activation, and dropout (<italic toggle="yes">P</italic> = .1). The final layer outputs 2D class logits per segment, followed by a softmax for binary classification.</p></caption><graphic xmlns:xlink="http://www.w3.org/1999/xlink" position="float" orientation="portrait" xlink:href="bbaf713f1.jpg"><?image-name bbaf713f1.jpg?><?image-size 27674?><?image-md5 35ecd57df3cbd5e04de860f5923eeaba?><?image-image-server-status LOAD_COMPLETED?><?image-original-height 951?><?image-original-width 3662?><?image-scaled-height 190?><?image-scaled-width 732?><?image-cloudpmc-urn urn:cdn:blobs/9241/12790621/35ecd57df3cb/bbaf713f1.jpg?><?thumb-name bbaf713f1.gif?><?thumb-size 3812?><?thumb-md5 dcdb3750a3db93d883686ac34682f5cd?><?thumb-image-server-status NEVER_LOAD?><?thumb-scaled-height 52?><?thumb-scaled-width 200?><?thumb-cloudpmc-urn urn:cdn:blobs/9241/12790621/dcdb3750a3db/bbaf713f1.gif?><alt-text>Alt Text: A structure of CoBRA model.</alt-text></graphic></fig><p>The model training was executed using the PyTorch [<xref rid="ref18" ref-type="bibr">18</xref>] framework. The optimization process was executed employing the AdamW [<xref rid="ref19" ref-type="bibr">19</xref>] optimizer, with an initial learning rate set to 5 × 10<sup>−4</sup>. The cosine annealing method was employed for the purpose of learning rate scheduling. All experiments were run with a batch size of 4, a dropout probability of .1, for 100 epochs, and with the random seed fixed to 42 to ensure reproducibility.</p></sec><sec id="sec8"><title>RNA language models</title><p>Ten pre-trained RNA LMs were employed to pick the best model for residue-level binding site prediction. The models were pre-trained on diverse RNA types for various learning objectives, resulting in differences in embedding dimensionality and representational properties. <xref rid="TB1" ref-type="table">Table 1</xref> lists the RNA LMs, their pre-training targets, and the embedding size.</p><table-wrap position="float" id="TB1" orientation="portrait"><label>Table 1</label><caption><p>Description of RNA language models employed for CoBRA.</p></caption><table frame="hsides" rules="groups"><colgroup span="1"><col align="left" span="1"/><col align="left" span="1"/><col align="left" span="1"/></colgroup><thead><tr><th align="left" rowspan="1" colspan="1">Model name</th><th align="left" rowspan="1" colspan="1">Pre-training target</th><th align="left" rowspan="1" colspan="1">Embedding size</th></tr></thead><tbody><tr><td align="left" rowspan="1" colspan="1">ERNIE-RNA</td><td align="left" rowspan="1" colspan="1">Non-coding RNA</td><td align="left" rowspan="1" colspan="1">768</td></tr><tr><td align="left" rowspan="1" colspan="1">RiNALMo</td><td align="left" rowspan="1" colspan="1">Non-coding RNA</td><td align="left" rowspan="1" colspan="1">1280</td></tr><tr><td align="left" rowspan="1" colspan="1">RNABERT</td><td align="left" rowspan="1" colspan="1">Non-coding RNA</td><td align="left" rowspan="1" colspan="1">120</td></tr><tr><td align="left" rowspan="1" colspan="1">RNA-FM</td><td align="left" rowspan="1" colspan="1">Non-coding RNA</td><td align="left" rowspan="1" colspan="1">640</td></tr><tr><td align="left" rowspan="1" colspan="1">RNA-MSM</td><td align="left" rowspan="1" colspan="1">Non-coding RNA</td><td align="left" rowspan="1" colspan="1">768</td></tr><tr><td align="left" rowspan="1" colspan="1">SpliceBERT</td><td align="left" rowspan="1" colspan="1">Precursor mRNA</td><td align="left" rowspan="1" colspan="1">512</td></tr><tr><td align="left" rowspan="1" colspan="1">SpliceBERT-510</td><td align="left" rowspan="1" colspan="1">Precursor mRNA</td><td align="left" rowspan="1" colspan="1">512</td></tr><tr><td align="left" rowspan="1" colspan="1">SpliceBERT-H510</td><td align="left" rowspan="1" colspan="1">Precursor mRNA</td><td align="left" rowspan="1" colspan="1">512</td></tr><tr><td align="left" rowspan="1" colspan="1">UTRLM-MRL</td><td align="left" rowspan="1" colspan="1">mRNA 5′UTR</td><td align="left" rowspan="1" colspan="1">128</td></tr><tr><td align="left" rowspan="1" colspan="1">UTRLM-TE_EL</td><td align="left" rowspan="1" colspan="1">mRNA 5′UTR</td><td align="left" rowspan="1" colspan="1">128</td></tr></tbody></table></table-wrap><p>The models can be categorized into two groups based on pre-training sets. The first category is the models trained on non-coding RNAs (ncRNAs), composed of ERNIE-RNA [<xref rid="ref20" ref-type="bibr">20</xref>], RiNALMo [<xref rid="ref21" ref-type="bibr">21</xref>], RNABERT [<xref rid="ref22" ref-type="bibr">22</xref>], RNA-FM [<xref rid="ref23" ref-type="bibr">23</xref>], and RNA-MSM [<xref rid="ref24" ref-type="bibr">24</xref>]. This group captures generalizable structural patterns across various ncRNA families. ERNIE-RNA [<xref rid="ref20" ref-type="bibr">20</xref>], comprised of 12 Transformer blocks, is pre-trained via masked language modeling (MLM) on 20 million ncRNAs from RNAcentral with structural information. RiNALMo [<xref rid="ref21" ref-type="bibr">21</xref>] is also pre-trained on 36 million ncRNAs, employing MLM with 33 Transformer blocks. RNABERT [<xref rid="ref22" ref-type="bibr">22</xref>] adopts the pre-training BERT algorithm to 762 K ncRNAs. It also encodes the characteristics of the RNA family and structure. RNA-FM [<xref rid="ref23" ref-type="bibr">23</xref>] is built upon 12 bidirectional Transformer encoder blocks that produce an L × 640 embedding matrix for input length L, pre-trained on 23.7 million ncRNAs. RNA-MSM [<xref rid="ref24" ref-type="bibr">24</xref>] follows the MSA Transformer architecture with ten blocks to learn two-dimensional co-evolutionary signals from homologous multiple sequence alignments of 3932 Rfam families.</p><p>The members of the second class are pre-trained on pre-mRNA or mRNA untranslated (UTR) regions. SpliceBERT, SpliceBERT-510, SpliceBERT-H510 [<xref rid="ref25" ref-type="bibr">25</xref>], UTRLM-MRL, and UTRLM-TE_EL [<xref rid="ref26" ref-type="bibr">26</xref>] are members of the class. The category specializes in modeling sequence features in post-transcriptional regulatory regions. SpliceBERT [<xref rid="ref25" ref-type="bibr">25</xref>] is a BERT-based model with six Transformer encoder layers pre-trained on 2 million RNA sequences from 72 vertebrate species. Additionally, two variants of the model were also employed: SpliceBERT-510, an intermediate checkpoint, and SpliceBERT-H510, pre-trained exclusively on human data. UTR-LM [<xref rid="ref26" ref-type="bibr">26</xref>] is a six-layer Transformer encoder pre-trained via semi-supervised masked nucleotide reconstruction, 5′ UTR secondary-structure prediction, and minimum-free-energy regression on 5′ UTRs from multiple species. Two task-specific variants, UTR-LM–TE_EL (translation efficiency and mRNA expression-level prediction) and UTR-LM–MRL (mean ribosome loading prediction) were also used to generate embeddings.</p><p>All language models were used in their pre-trained form with frozen parameters; no fine-tuning or weight updates were performed during training. Each RNA sequence was transformed into a residue-level embedding sequence using the corresponding model, then padded to a fixed length of 161 residues before being fed into the MLP classifier. Embedding dimensionality varied across models, ranging from 120 to 1280. The predictive performance of different embedding types was compared to assess their impact on RNA–ligand binding site prediction. Additionally, a series of models was constructed and tested to connect the two LMs by concatenating their embeddings. Furthermore, one-hot encoding of bases was employed in order to provide a baseline.</p></sec><sec id="sec9"><title>Loss functions</title><p>Compared to the RNA sequence length, the proportion of binding nucleotides (positive class) is low, leading to a severe class imbalance problem. To resolve the issue and quantitatively evaluate how different training objectives affect performance, we employed six loss functions.</p><p>Each loss function represents a different optimization strategy. Binary cross-entropy (BCE) serves as the conventional baseline for binary classification, encouraging predicted probabilities to converge to the ground truth. Class-balanced focal loss [<xref rid="ref27" ref-type="bibr">27</xref>] and Tversky loss [<xref rid="ref28" ref-type="bibr">28</xref>] are designed to address the class imbalance. The class-balanced focal loss emphasizes hard-to-classify samples by assigning them higher weights. In contrast, the Tversky loss applies asymmetric weighting to false positives and false negatives. By giving different penalties, the loss function tries to improve model sensitivity for an imbalanced dataset. Dice loss [<xref rid="ref29" ref-type="bibr">29</xref>] and Lovász hinge loss [<xref rid="ref30" ref-type="bibr">30</xref>] focus on residue-level spatial alignment and structural consistency. Dice loss maximizes the overlap between predicted and true positive regions, whereas Lovász hinge loss optimizes the Intersection over Union directly.</p><p>Lastly, we introduce a composite loss that combines triplet center loss (TCL) [<xref rid="ref31" ref-type="bibr">31</xref>] and class-balanced focal loss to simultaneously improve class separation in the embedding space and address the class imbalance. This design is inspired by Wang <italic toggle="yes">et al.</italic> [<xref rid="ref32" ref-type="bibr">32</xref>], demonstrated strong performance in protein-small molecule binding site prediction using a combination of TCL and class-balanced focal loss to enhance both feature discrimination and class imbalance handling. The total loss is defined as shown in Equation <xref rid="deqn01" ref-type="disp-formula">1</xref>.</p><disp-formula id="deqn01">
<label>(1)</label>
<tex-math notation="LaTeX" id="DmEquation1"><?equation-image-name DmEquation1.gif?><?equation-image-status READY?><?equation-image-md5 5d45f48e33a887c355598fc2393031b3?><?equation-image-cloudpmc-urn urn:cdn:blobs/9241/12790621/5d45f48e33a8/DmEquation1.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym}
\usepackage{amsfonts}
\usepackage{amssymb}
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
\begin{equation*} {L}_{\text{total}}={L}_{\text{focal}}+\lambda \cdotp{L}_{\text{tcl}} \end{equation*}\end{document}</tex-math>
</disp-formula><p>
<italic toggle="yes">L</italic>
<sub>focal</sub> and <italic toggle="yes">L</italic><sub>tcl</sub> are class-balanced focal loss and TCL loss, respectively. <italic toggle="yes">λ</italic> is used as a weight to balance between the losses. <italic toggle="yes">L</italic><sub>focal</sub>, a class-balanced focal loss, assigns higher weights to address class imbalance, as shown in Equation (<xref rid="deqn02" ref-type="disp-formula">2</xref>).</p><disp-formula id="deqn02">
<label>(2)</label>
<tex-math notation="LaTeX" id="DmEquation2"><?equation-image-name DmEquation2.gif?><?equation-image-status READY?><?equation-image-md5 ea9bfbd766671a516042cdab97fba0d9?><?equation-image-cloudpmc-urn urn:cdn:blobs/9241/12790621/ea9bfbd76667/DmEquation2.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym}
\usepackage{amsfonts}
\usepackage{amssymb}
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
\begin{equation*} {L}_{\text{focal}}=-{E}_n\sum_{i=1}^C{\left(1-{p}_i^t\right)}^{\gamma}\log \left({p}_i^t\right),\text{where}\ {E}_n=\frac{1-\beta }{1-{\beta}^{n_y}} \end{equation*}\end{document}</tex-math>
</disp-formula><p>
<italic toggle="yes">p<sub>i</sub><sup>t</sup></italic> denotes the predicted probability of the true class <italic toggle="yes">t</italic>, and <italic toggle="yes">(1-p<sub>i</sub><sup>t</sup>)<sup>g</sup></italic> serves as a modulating factor that down-weights well-classified samples and focuses the training on hard examples. The focusing parameter <italic toggle="yes">g</italic> controls the relative importance of easy and hard samples. The term <italic toggle="yes">E<sub>n</sub></italic>, represents the effective number of samples for each class and balances the impact of class frequency.</p><p>TCL learns a center vector for each class and encourages embeddings of the same class to cluster together while enforcing a margin-based separation between different classes (Equation <xref rid="deqn03" ref-type="disp-formula">3</xref>).</p><disp-formula id="deqn03">
<label>(3)</label>
<tex-math notation="LaTeX" id="DmEquation3"><?equation-image-name DmEquation3.gif?><?equation-image-status READY?><?equation-image-md5 ade959fd2fdd2ac1025a678ac64c1420?><?equation-image-cloudpmc-urn urn:cdn:blobs/9241/12790621/ade959fd2fdd/DmEquation3.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym}
\usepackage{amsfonts}
\usepackage{amssymb}
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
\begin{equation*} {L}_{\text{tcl}}=\sum_{i=1}^M\max \left(D\left({f}_i,{c}_{y_i}\right)+m-D\left({f}_i,{c}_{1-{y}_i}\right),0\right)\kern0.5em \end{equation*}\end{document}</tex-math>
</disp-formula><p>Here, <italic toggle="yes">f</italic> is the embedding vector of the input, <italic toggle="yes">c<sub>y</sub></italic> and <italic toggle="yes">c<sub>1 − y</sub></italic> denote the center vectors of the true and opposite classes, respectively, and m is the margin. The hyperparameters were set as g = 3, b = 0.999, l = 0.2, and m = 4, following the optimal configuration reported by Wang <italic toggle="yes">et al.</italic> [<xref rid="ref32" ref-type="bibr">32</xref>].</p></sec><sec id="sec10"><title>Evaluation metrics</title><p>All metrics are calculated based on the values of true positive (TP), false positive (FP), true negative (TN), and false negative (FN) in the confusion matrix. Precision is the ratio of TPs to all predicted positives (TP + FP), while recall is the percentage of TPs to all actual positives (TP + FN). F1-score is a harmonized mean of precision and recall that evaluates the balance between the two metrics. The Matthews correlation coefficient (MCC), a class imbalance robust metric, includes all elements of the confusion matrix, calculated as follows:</p><disp-formula id="deqn04">
<label>(4)</label>
<tex-math notation="LaTeX" id="DmEquation4"><?equation-image-name DmEquation4.gif?><?equation-image-status READY?><?equation-image-md5 5b6e9f5bfe15dc2245294145a36ba2cf?><?equation-image-cloudpmc-urn urn:cdn:blobs/9241/12790621/5b6e9f5bfe15/DmEquation4.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym}
\usepackage{amsfonts}
\usepackage{amssymb}
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
\begin{equation*} \text{MCC}=\frac{\text{TP}\times \text{TN}-\text{FP}\times \text{FN}}{\sqrt{\left(\text{TP}+\text{FP}\right)\left(\text{TP}+\text{FN}\right)\left(\text{TN}+\text{FP}\right)\left(\text{TN}+\text{FN}\right)}} \end{equation*}\end{document}</tex-math>
</disp-formula><p>The value of MCC can range from +1 (perfect prediction) to −1 (perfect mismatch).</p><p>The area under the receiver operating characteristic curve (AUROC) quantifies the model’s discrimination ability by measuring the area under the ROC curve (TP rate versus FP rate), and area under the precision–recall curve (AUPRC) measures the area under the precision–recall curve, summarizing the trade-off between precision and recall across all thresholds.</p></sec><sec id="sec11"><title>Laplacian-based curvature for RNA 3D structure analysis</title><p>To analyze and characterize the three-dimensional RNA structures quantitatively, the Laplacian-based curvature descriptor (LN) [<xref rid="ref12" ref-type="bibr">12</xref>] is employed. This metric quantifies the curvature at each nucleotide by measuring the deviation of its C3′ atom coordinates from the relative positions of surrounding residues. It thereby captures local structural distortion and reflects how naturally a nucleotide is embedded within the overall structure.</p><p>For each nucleotide, a coordinate vector was defined by the C3′ atom position. Then a Gaussian kernel-based weighting function was constructed from the all residue pairwise Euclidean distance matrix following Equation <xref rid="deqn05" ref-type="disp-formula">5</xref>.</p><disp-formula id="deqn05">
<label>(5)</label>
<tex-math notation="LaTeX" id="DmEquation5"><?equation-image-name DmEquation5.gif?><?equation-image-status READY?><?equation-image-md5 ad5886917029007e553a5cda35bd8687?><?equation-image-cloudpmc-urn urn:cdn:blobs/9241/12790621/ad5886917029/DmEquation5.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym}
\usepackage{amsfonts}
\usepackage{amssymb}
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
\begin{equation*} {\Omega}_{ij}\left(\delta \right)=\left\{\begin{array}{@{}cc}\exp \left(-\frac{{\left\Vert{p}_i-{p}_j\right\Vert}^2}{\delta^2}\right)&amp; \left|i-j\right|&gt;1\\{}0&amp; \text{Otherwise}\end{array}\right. \end{equation*}\end{document}</tex-math>
</disp-formula><p>Here <italic toggle="yes">d</italic> is a single-scale parameter, set as the median of all pairwise distances between nucleotides in the structure. <italic toggle="yes">p<sub>i</sub></italic> and <italic toggle="yes">p<sub>j</sub></italic> denote the Euclidean coordinates of the C3′ atoms of nucleotides 𝑖 and 𝑗. This choice of scale balances local and global shape sensitivity, avoiding excessive localization or over-smoothing.</p><p>Using the weight function, the Laplacian norm value LN<italic toggle="yes"><sub>i</sub></italic> for nucleotide <italic toggle="yes">i</italic> is defined as Equation <xref rid="deqn06" ref-type="disp-formula">6</xref>.</p><disp-formula id="deqn06">
<label>(6)</label>
<tex-math notation="LaTeX" id="DmEquation6"><?equation-image-name DmEquation6.gif?><?equation-image-status READY?><?equation-image-md5 7c3cf146a5d07bb8650dfe3eb3f8f52a?><?equation-image-cloudpmc-urn urn:cdn:blobs/9241/12790621/7c3cf146a5d0/DmEquation6.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym}
\usepackage{amsfonts}
\usepackage{amssymb}
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
\begin{equation*} {LN}_i\left(\delta \right)=\left\Vert{p}_i-\frac{\sum_{j,\left|i-j\right|&gt;1}{p}_j{\Omega}_{ij}\left(\delta \right)}{\sum_{j,\left|i-j\right|&gt;1}{\Omega}_{ij}\left(\delta \right)}\right\Vert \end{equation*}\end{document}</tex-math>
</disp-formula><p>This value reflects how far nucleotide <italic toggle="yes">i</italic> lies from its Gaussian-weighted center, capturing the relative positional characteristics within the RNA structure. Higher LN values indicate convex or protruding regions, while lower values suggest concave or densely packed areas. The computed LN values were utilized for analysis and structural visualization.</p></sec><sec id="sec12"><title>Evaluating computational cost with comparison to other machine learning</title><p>To evaluate the computational cost, all sequences were embedded using the frozen RiNALMo model on an NVIDIA RTX A6000 GPU. The resulting 1280-dimensional embeddings were used to compare the cost of logistic regression (LR) and random forest (RF). The hyperparameter settings for LR and RF are provided in <xref rid="sup1" ref-type="supplementary-material">Supplementary Table S1</xref>.</p></sec></sec><sec id="sec13"><title>Results</title><sec id="sec14"><title>Overall performance comparison and model selection</title><p>A total of 60 models were evaluated and compared, incorporating 10 RNA LM embeddings and 6 loss functions. <xref rid="TB2" ref-type="table">Table 2</xref> lists the top 10 MCC models and the best one-hot encoding model, benchmarked on the test set. <xref rid="sup1" ref-type="supplementary-material">Supplementary Table S2</xref> showed the performance of individual models.</p><table-wrap position="float" id="TB2" orientation="portrait"><label>Table 2</label><caption><p>Performance of the top 10 MCC models and a baseline using one-hot encoding on the test set.<xref rid="tblfn1" ref-type="table-fn"><sup>a</sup></xref></p></caption><table frame="hsides" rules="groups"><colgroup span="1"><col align="left" span="1"/><col align="left" span="1"/><col align="left" span="1"/><col align="left" span="1"/><col align="left" span="1"/><col align="left" span="1"/><col align="left" span="1"/></colgroup><thead><tr><th align="left" rowspan="1" colspan="1">Language model</th><th align="left" rowspan="1" colspan="1">Loss function</th><th align="left" rowspan="1" colspan="1">MCC</th><th align="left" rowspan="1" colspan="1">AUROC</th><th align="left" rowspan="1" colspan="1">AUPRC</th><th align="left" rowspan="1" colspan="1">Precision</th><th align="left" rowspan="1" colspan="1">Recall</th></tr></thead><tbody><tr><td align="left" rowspan="1" colspan="1">ERNIE-RNA</td><td align="left" rowspan="1" colspan="1">TCL focal</td><td align="left" rowspan="1" colspan="1">
<bold>0.657</bold>
</td><td align="left" rowspan="1" colspan="1">0.868</td><td align="left" rowspan="1" colspan="1">
<bold>0.817</bold>
</td><td align="left" rowspan="1" colspan="1">
<bold>0.787</bold>
</td><td align="left" rowspan="1" colspan="1">0.718</td></tr><tr><td align="left" rowspan="1" colspan="1">RNA-FM</td><td align="left" rowspan="1" colspan="1">TCL focal</td><td align="left" rowspan="1" colspan="1">0.641</td><td align="left" rowspan="1" colspan="1">0.876</td><td align="left" rowspan="1" colspan="1">0.806</td><td align="left" rowspan="1" colspan="1">0.755</td><td align="left" rowspan="1" colspan="1">0.732</td></tr><tr><td align="left" rowspan="1" colspan="1">RiNALMo</td><td align="left" rowspan="1" colspan="1">BCE</td><td align="left" rowspan="1" colspan="1">0.629</td><td align="left" rowspan="1" colspan="1">0.881</td><td align="left" rowspan="1" colspan="1">0.816</td><td align="left" rowspan="1" colspan="1">0.737</td><td align="left" rowspan="1" colspan="1">0.738</td></tr><tr><td align="left" rowspan="1" colspan="1">ERNIE-RNA</td><td align="left" rowspan="1" colspan="1">Focal</td><td align="left" rowspan="1" colspan="1">0.626</td><td align="left" rowspan="1" colspan="1">0.874</td><td align="left" rowspan="1" colspan="1">0.808</td><td align="left" rowspan="1" colspan="1">0.741</td><td align="left" rowspan="1" colspan="1">0.726</td></tr><tr><td align="left" rowspan="1" colspan="1">ERNIE-RNA</td><td align="left" rowspan="1" colspan="1">BCE</td><td align="left" rowspan="1" colspan="1">0.625</td><td align="left" rowspan="1" colspan="1">0.863</td><td align="left" rowspan="1" colspan="1">0.798</td><td align="left" rowspan="1" colspan="1">0.750</td><td align="left" rowspan="1" colspan="1">0.714</td></tr><tr><td align="left" rowspan="1" colspan="1">RiNALMo</td><td align="left" rowspan="1" colspan="1">Lovasz hinge</td><td align="left" rowspan="1" colspan="1">0.618</td><td align="left" rowspan="1" colspan="1">0.874</td><td align="left" rowspan="1" colspan="1">0.779</td><td align="left" rowspan="1" colspan="1">0.731</td><td align="left" rowspan="1" colspan="1">0.727</td></tr><tr><td align="left" rowspan="1" colspan="1">RNA-FM</td><td align="left" rowspan="1" colspan="1">Focal</td><td align="left" rowspan="1" colspan="1">0.615</td><td align="left" rowspan="1" colspan="1">
<bold>0.887</bold>
</td><td align="left" rowspan="1" colspan="1">0.816</td><td align="left" rowspan="1" colspan="1">0.737</td><td align="left" rowspan="1" colspan="1">0.714</td></tr><tr><td align="left" rowspan="1" colspan="1">RNA-FM</td><td align="left" rowspan="1" colspan="1">BCE</td><td align="left" rowspan="1" colspan="1">0.613</td><td align="left" rowspan="1" colspan="1">0.881</td><td align="left" rowspan="1" colspan="1">0.808</td><td align="left" rowspan="1" colspan="1">0.706</td><td align="left" rowspan="1" colspan="1">
<bold>0.753</bold>
</td></tr><tr><td align="left" rowspan="1" colspan="1">SpliceBERT-510</td><td align="left" rowspan="1" colspan="1">BCE</td><td align="left" rowspan="1" colspan="1">0.613</td><td align="left" rowspan="1" colspan="1">0.853</td><td align="left" rowspan="1" colspan="1">0.785</td><td align="left" rowspan="1" colspan="1">0.740</td><td align="left" rowspan="1" colspan="1">0.705</td></tr><tr><td align="left" rowspan="1" colspan="1">RiNALMo</td><td align="left" rowspan="1" colspan="1">TCL focal</td><td align="left" rowspan="1" colspan="1">0.609</td><td align="left" rowspan="1" colspan="1">0.876</td><td align="left" rowspan="1" colspan="1">0.802</td><td align="left" rowspan="1" colspan="1">0.713</td><td align="left" rowspan="1" colspan="1">0.736</td></tr><tr><td align="left" rowspan="1" colspan="1">One-hot</td><td align="left" rowspan="1" colspan="1">Tversky</td><td align="left" rowspan="1" colspan="1">0.099</td><td align="left" rowspan="1" colspan="1">0.568</td><td align="left" rowspan="1" colspan="1">0.329</td><td align="left" rowspan="1" colspan="1">0.336</td><td align="left" rowspan="1" colspan="1">0.573</td></tr></tbody></table><table-wrap-foot><fn id="tblfn1"><p>
<sup>a</sup>The highest value for each metric among the ten models is highlighted in bold.</p></fn></table-wrap-foot></table-wrap><p>All the top 10 models demonstrated superior performance to the baseline model (MCC: 0.099, AUROC: 0.568), confirming that the performance improvements of CoBRA primarily originate from the contextual representations learned by RNA language models. The highest MCC is obtained by combining ERNIE-RNA with TCL focal as the language model and loss function, respectively. The model has also been determined to be second in terms of AUPRC and precision among the 60 models considered. It was observed that ERNIE-RNA, RNA-FM, and RiNALMo emerged as top contenders, appearing in the top ten MCC models on three separate occasions. The three LMs were pre-trained from ncRNAs. Conversely, the LMs pre-trained on mRNA-UTR regions are not included among the top 10, except for SpliceBERT-510. The comparatively diminished performance of mRNA-pretrained models might be due to their constrained exposure to a variety of RNA structures, which could result in diminished generalizability in binding site prediction when compared to ncRNA-pretrained counterparts. A survey of the top 10 MCC models reveals that 9 of them utilize entropy-like loss functions, including focal, TCL focal, and BCE. Except for the three aforementioned functions, the Lovasz hinge with RiNALMo is ranked sixth in terms of MCC.</p><p>The distribution of MCC, AUPRC, and AUROC for all 60 models is illustrated in <xref rid="f2" ref-type="fig">Fig. 2</xref> and <xref rid="sup1" ref-type="supplementary-material">Supplementary Table S3</xref> across the LMs and loss functions. As observed by the top 10 MCC models, RiNALMo, ERNIE-RNA, and RNA-FM, which were pre-trained on ncRNAs, exhibited the highest average MCC, AUROC, AUPRC, and precision values (<xref rid="f2" ref-type="fig">Fig. 2A</xref>). The five LMs, trained on mRNA-UTR regions, demonstrated consistent performance, with MCC ranging from 0.292 to 0.346. Conversely, the performances of LMs trained on ncRNAs exhibit significant variability. As observed in the top 10 models, ERNIE-RNA, RiNALMo, and RNA-FM, which were trained on more than 20 million sequences, exhibited superior performance to mRNA-UTR models, while RNABERT and RNA-MSM showed lower MCC values than UTR models. In terms of MCC, the best model using RNABERT (0.331) performed worse than the worst model using RiNALMo (0.372).</p><fig position="float" id="f2" orientation="portrait"><label>Figure 2</label><caption><p>Distribution of MCC (left) AUPRC (middle), and AUROC (right) across all RNA LMs (A) and loss functions (B).</p></caption><graphic xmlns:xlink="http://www.w3.org/1999/xlink" position="float" orientation="portrait" xlink:href="bbaf713f2.jpg"><?image-name bbaf713f2.jpg?><?image-size 44495?><?image-md5 98aa8f8c9c0c46e7a61c935eb14eb331?><?image-image-server-status LOAD_COMPLETED?><?image-original-height 1734?><?image-original-width 4000?><?image-scaled-height 347?><?image-scaled-width 800?><?image-cloudpmc-urn urn:cdn:blobs/9241/12790621/98aa8f8c9c0c/bbaf713f2.jpg?><?thumb-name bbaf713f2.gif?><?thumb-size 4452?><?thumb-md5 4524e87a6e02df8e0241fbe0bc7b5503?><?thumb-image-server-status NEVER_LOAD?><?thumb-scaled-height 80?><?thumb-scaled-width 184?><?thumb-cloudpmc-urn urn:cdn:blobs/9241/12790621/4524e87a6e02/bbaf713f2.gif?><alt-text>Alt Text: Bar plots compare the performances across the 6 RNA LMs and the 10 loss functions.</alt-text></graphic></fig><p>With regard to loss functions, models that have been trained with BCE and focal loss have demonstrated a consistent and superior performance across key evaluation metrics in comparison to other loss functions. Conversely, models employing dice loss and Lovasz hinge loss exhibited unstable convergence and suboptimal performance in most cases (<xref rid="f2" ref-type="fig">Fig. 2B</xref>).</p><p>The LMs with the top three MCC values, ERNIE-RNA, RNA-FM, and RiNALMo, were subsequently selected for the concatenation experiment (<xref rid="sup1" ref-type="supplementary-material">Supplementary Table S4</xref>). The combination of ERNIE-RNA and RNA-FM yielded the highest MCC among the concatenated models. However, none of the models performed MCC higher than the best model of single LM embedding. This implies that the embeddings might contain analogous information, rendering their combination ineffective.</p><p>Based on the single-embedding models from the main experiments, a selection of the top five MCC configurations was made, and subsequent benchmarking was conducted on the external validation sets.</p></sec><sec id="sec15"><title>Structure-based split dataset evaluation</title><p>To assess the generalizability of the models, the sixty models were retrained and evaluated using the RNABind structure-based split dataset. This ensures strict non-redundancy between the training and the test sets at both levels. The results of the top five models are given in <xref rid="sup1" ref-type="supplementary-material">Supplementary Table S5</xref>. Compared to the methods that combine sequence information as LMs and structural information, CoBRA demonstrated lower AUROC values ranging from 0.605 to 0.657, while RNABind, RLBind, Ret, and MVRBind reported AUROC values ranging from 0.671 to 0.776. On the other hand, CoBRA and RNABind demonstrated similar performance in terms of AUPRC.</p></sec><sec id="sec16"><title>Comparison with other existing methods on benchmark sets</title><p>We conducted a benchmarking study of the top five MCC models on various test sets composed of RNA-compound crystal structures. RB9, JL10, TL12, and TE18. <xref rid="TB3" ref-type="table">Table 3</xref> presents the performance of the models and other RNA-compound binding site prediction methods. The CoBRA models are designated CoBRA-M1, M2, and so forth, following their MCC ranks.</p><table-wrap position="float" id="TB3" orientation="portrait"><label>Table 3</label><caption><p>Comparison of CoBRA model performances on various test sets.<xref rid="tblfn2" ref-type="table-fn"><sup>a</sup></xref></p></caption><table frame="hsides" rules="groups"><colgroup span="1"><col align="left" span="1"/><col align="left" span="1"/><col align="left" span="1"/><col align="left" span="1"/><col align="left" span="1"/><col align="left" span="1"/></colgroup><thead><tr><td rowspan="1" colspan="1"/><th align="left" rowspan="1" colspan="1">M1</th><th align="left" rowspan="1" colspan="1">M2</th><th align="left" rowspan="1" colspan="1">M3</th><th align="left" rowspan="1" colspan="1">M4</th><th align="left" rowspan="1" colspan="1">M5</th></tr><tr><th align="left" rowspan="1" colspan="1">Language model</th><th align="left" rowspan="1" colspan="1">ERNIE-RNA</th><th align="left" rowspan="1" colspan="1">RNA-FM</th><th align="left" rowspan="1" colspan="1">RiNALMo</th><th align="left" rowspan="1" colspan="1">ERNIE-RNA</th><th align="left" rowspan="1" colspan="1">ERNIE-RNA</th></tr><tr><th align="left" rowspan="1" colspan="1">Loss function</th><th align="left" rowspan="1" colspan="1">TCL focal</th><th align="left" rowspan="1" colspan="1">TCL focal</th><th align="left" rowspan="1" colspan="1">BCE</th><th align="left" rowspan="1" colspan="1">Focal</th><th align="left" rowspan="1" colspan="1">BCE</th></tr></thead><tbody><tr><td align="left" rowspan="1" colspan="1">RB9</td><td rowspan="1" colspan="1"/><td rowspan="1" colspan="1"/><td rowspan="1" colspan="1"/><td rowspan="1" colspan="1"/><td rowspan="1" colspan="1"/></tr><tr><td align="left" rowspan="1" colspan="1"> MCC</td><td align="left" rowspan="1" colspan="1">0.585</td><td align="left" rowspan="1" colspan="1">
<bold>0.593</bold>
</td><td align="left" rowspan="1" colspan="1">0.557</td><td align="left" rowspan="1" colspan="1">0.538</td><td align="left" rowspan="1" colspan="1">0.559</td></tr><tr><td align="left" rowspan="1" colspan="1"> AUROC</td><td align="left" rowspan="1" colspan="1">0.829</td><td align="left" rowspan="1" colspan="1">
<bold>0.860</bold>
</td><td align="left" rowspan="1" colspan="1">0.844</td><td align="left" rowspan="1" colspan="1">0.845</td><td align="left" rowspan="1" colspan="1">0.855</td></tr><tr><td align="left" rowspan="1" colspan="1"> Precision</td><td align="left" rowspan="1" colspan="1">
<bold>0.789</bold>
</td><td align="left" rowspan="1" colspan="1">0.770</td><td align="left" rowspan="1" colspan="1">0.750</td><td align="left" rowspan="1" colspan="1">0.714</td><td align="left" rowspan="1" colspan="1">0.743</td></tr><tr><td align="left" rowspan="1" colspan="1"> Recall</td><td align="left" rowspan="1" colspan="1">0.643</td><td align="left" rowspan="1" colspan="1">
<bold>0.682</bold>
</td><td align="left" rowspan="1" colspan="1">0.650</td><td align="left" rowspan="1" colspan="1">0.669</td><td align="left" rowspan="1" colspan="1">0.662</td></tr><tr><td align="left" colspan="6" rowspan="1">JL10</td></tr><tr><td align="left" rowspan="1" colspan="1"> MCC</td><td align="left" rowspan="1" colspan="1">0.272</td><td align="left" rowspan="1" colspan="1">
<bold>0.331</bold>
</td><td align="left" rowspan="1" colspan="1">0.293</td><td align="left" rowspan="1" colspan="1">0.265</td><td align="left" rowspan="1" colspan="1">0.292</td></tr><tr><td align="left" rowspan="1" colspan="1"> AUROC</td><td align="left" rowspan="1" colspan="1">0.721</td><td align="left" rowspan="1" colspan="1">
<bold>0.747</bold>
</td><td align="left" rowspan="1" colspan="1">0.739</td><td align="left" rowspan="1" colspan="1">0.713</td><td align="left" rowspan="1" colspan="1">0.720</td></tr><tr><td align="left" rowspan="1" colspan="1"> Precision</td><td align="left" rowspan="1" colspan="1">0.590</td><td align="left" rowspan="1" colspan="1">
<bold>0.656</bold>
</td><td align="left" rowspan="1" colspan="1">0.582</td><td align="left" rowspan="1" colspan="1">0.563</td><td align="left" rowspan="1" colspan="1">
<bold>0.656</bold>
</td></tr><tr><td align="left" rowspan="1" colspan="1"> Recall</td><td align="left" rowspan="1" colspan="1">0.463</td><td align="left" rowspan="1" colspan="1">0.453</td><td align="left" rowspan="1" colspan="1">
<bold>0.531</bold>
</td><td align="left" rowspan="1" colspan="1">0.516</td><td align="left" rowspan="1" colspan="1">0.375</td></tr><tr><td align="left" colspan="6" rowspan="1">TL12</td></tr><tr><td align="left" rowspan="1" colspan="1"> MCC</td><td align="left" rowspan="1" colspan="1">
<bold>0.575</bold>
</td><td align="left" rowspan="1" colspan="1">0.521</td><td align="left" rowspan="1" colspan="1">0.546</td><td align="left" rowspan="1" colspan="1">0.520</td><td align="left" rowspan="1" colspan="1">0.499</td></tr><tr><td align="left" rowspan="1" colspan="1"> AUROC</td><td align="left" rowspan="1" colspan="1">0.821</td><td align="left" rowspan="1" colspan="1">0.800</td><td align="left" rowspan="1" colspan="1">0.816</td><td align="left" rowspan="1" colspan="1">0.819</td><td align="left" rowspan="1" colspan="1">
<bold>0.824</bold>
</td></tr><tr><td align="left" rowspan="1" colspan="1"> Precision</td><td align="left" rowspan="1" colspan="1">
<bold>0.859</bold>
</td><td align="left" rowspan="1" colspan="1">0.808</td><td align="left" rowspan="1" colspan="1">0.816</td><td align="left" rowspan="1" colspan="1">0.792</td><td align="left" rowspan="1" colspan="1">0.785</td></tr><tr><td align="left" rowspan="1" colspan="1"> Recall</td><td align="left" rowspan="1" colspan="1">
<bold>0.589</bold>
</td><td align="left" rowspan="1" colspan="1">0.570</td><td align="left" rowspan="1" colspan="1">0.599</td><td align="left" rowspan="1" colspan="1">0.589</td><td align="left" rowspan="1" colspan="1">0.565</td></tr><tr><td align="left" colspan="6" rowspan="1">TE18</td></tr><tr><td align="left" rowspan="1" colspan="1"> MCC</td><td align="left" rowspan="1" colspan="1">0.103</td><td align="left" rowspan="1" colspan="1">0.114</td><td align="left" rowspan="1" colspan="1">
<bold>0.190</bold>
</td><td align="left" rowspan="1" colspan="1">0.162</td><td align="left" rowspan="1" colspan="1">0.176</td></tr><tr><td align="left" rowspan="1" colspan="1"> AUROC</td><td align="left" rowspan="1" colspan="1">0.580</td><td align="left" rowspan="1" colspan="1">0.594</td><td align="left" rowspan="1" colspan="1">
<bold>0.647</bold>
</td><td align="left" rowspan="1" colspan="1">0.634</td><td align="left" rowspan="1" colspan="1">0.626</td></tr><tr><td align="left" rowspan="1" colspan="1"> Precision</td><td align="left" rowspan="1" colspan="1">0.504</td><td align="left" rowspan="1" colspan="1">0.473</td><td align="left" rowspan="1" colspan="1">
<bold>0.562</bold>
</td><td align="left" rowspan="1" colspan="1">0.525</td><td align="left" rowspan="1" colspan="1">0.537</td></tr><tr><td align="left" rowspan="1" colspan="1"> Recall</td><td align="left" rowspan="1" colspan="1">0.253</td><td align="left" rowspan="1" colspan="1">
<bold>0.465</bold>
</td><td align="left" rowspan="1" colspan="1">0.357</td><td align="left" rowspan="1" colspan="1">0.390</td><td align="left" rowspan="1" colspan="1">0.394</td></tr><tr><td align="left" colspan="6" rowspan="1">Average</td></tr><tr><td align="left" rowspan="1" colspan="1"> MCC</td><td align="left" rowspan="1" colspan="1">0.384</td><td align="left" rowspan="1" colspan="1">0.390</td><td align="left" rowspan="1" colspan="1">
<bold>0.397</bold>
</td><td align="left" rowspan="1" colspan="1">0.371</td><td align="left" rowspan="1" colspan="1">0.382</td></tr><tr><td align="left" rowspan="1" colspan="1"> AUROC</td><td align="left" rowspan="1" colspan="1">0.738</td><td align="left" rowspan="1" colspan="1">0.750</td><td align="left" rowspan="1" colspan="1">
<bold>0.762</bold>
</td><td align="left" rowspan="1" colspan="1">0.753</td><td align="left" rowspan="1" colspan="1">0.756</td></tr><tr><td align="left" rowspan="1" colspan="1"> Precision</td><td align="left" rowspan="1" colspan="1">
<bold>0.686</bold>
</td><td align="left" rowspan="1" colspan="1">0.677</td><td align="left" rowspan="1" colspan="1">0.678</td><td align="left" rowspan="1" colspan="1">0.649</td><td align="left" rowspan="1" colspan="1">0.680</td></tr><tr><td align="left" rowspan="1" colspan="1"> Recall</td><td align="left" rowspan="1" colspan="1">0.487</td><td align="left" rowspan="1" colspan="1">
<bold>0.543</bold>
</td><td align="left" rowspan="1" colspan="1">0.534</td><td align="left" rowspan="1" colspan="1">0.541</td><td align="left" rowspan="1" colspan="1">0.499</td></tr></tbody></table><table-wrap-foot><fn id="tblfn2"><p>
<sup>a</sup>The top five MCC models were selected for validation. The highest value for each metric is highlighted in bold.</p></fn></table-wrap-foot></table-wrap><p>While the CoBRA-M1, ERNIE-RNA combined with TCL focal model demonstrated the highest MCC, AUPRC, and precision among the five models evaluated on the test set, its performance was not optimal on the four benchmark sets. Conversely, CoBRA-M3, utilizing RiNALMo as the RNA LM and BCE as the loss function, exhibited the highest performance in MCC and AUROC on average of the four test sets. For precision and recall, M1 and M2 performed best, respectively.</p><p>With regard to the MCC from individual benchmark sets, M2 has the highest value for JL10 and RB9, whereas M1 shows the best MCC for TL12. While M3 is only ranked top for TE18 among the five CoBRA models, it is ranked second for JL10, a set with a high structural complexity, and TL12, a set with a low structural complexity. This suggests that M3 has greater generalizability for RNA-compound binding site prediction problems than other models. Consequently, M3 is selected as a representative model for CoBRA, and the results are subjected to further analysis.</p><p>A comparative analysis of RNA-compound binding site prediction methods revealed that CoBRA exhibited superior performance, with the exception of TE18 (<xref rid="TB4" ref-type="table">Table 4</xref>). On average, CoBRA exhibited a 15.4%, 9.2%, and 33.8% improvement in MCC, AUROC, and recall, respectively, when compared to ZeSTa, a state-of-the-art program with structural information. In particular, on datasets with relatively simpler structures, such as RB9 and TL12, CoBRA also outperformed other methods, with the exception of RB9 precision and AUROC. In a similar vein, the CoBRA algorithm demonstrated the most optimal performance in all metrics for the highly complex structure set, JL10. Conversely, the TE18 dataset exhibited a decline in performance, with an MCC reduction of 41.90% in comparison to ZeSTa.</p><table-wrap position="float" id="TB4" orientation="portrait"><label>Table 4</label><caption><p>Comparison of performances with other RNA-ligand binding site prediction programs.<xref rid="tblfn3" ref-type="table-fn"><sup>a</sup></xref></p></caption><table frame="hsides" rules="groups"><colgroup span="1"><col align="left" span="1"/><col align="left" span="1"/><col align="left" span="1"/><col align="left" span="1"/><col align="left" span="1"/><col align="left" span="1"/><col align="left" span="1"/><col align="left" span="1"/><col align="left" span="1"/><col align="left" span="1"/></colgroup><thead><tr><td rowspan="1" colspan="1"/><th align="left" rowspan="1" colspan="1">Rsite</th><th align="left" rowspan="1" colspan="1">Rsite2</th><th align="left" rowspan="1" colspan="1">RNAsite</th><th align="left" rowspan="1" colspan="1">RBind</th><th align="left" rowspan="1" colspan="1">RLBind</th><th align="left" rowspan="1" colspan="1">ZeSTa</th><th align="left" rowspan="1" colspan="1">RNABind</th><th align="left" rowspan="1" colspan="1">MVRBind</th><th align="left" rowspan="1" colspan="1">CoBRA</th></tr></thead><tbody><tr><td align="left" colspan="10" rowspan="1">RB9</td></tr><tr><td align="left" rowspan="1" colspan="1"> MCC</td><td align="left" rowspan="1" colspan="1">0.159</td><td align="left" rowspan="1" colspan="1">0.072</td><td align="left" rowspan="1" colspan="1">0.426</td><td align="left" rowspan="1" colspan="1">0.278</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">0.398</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">
<bold>0.557</bold>
</td></tr><tr><td align="left" rowspan="1" colspan="1"> AUROC</td><td align="left" rowspan="1" colspan="1">0.575</td><td align="left" rowspan="1" colspan="1">0.528</td><td align="left" rowspan="1" colspan="1">0.790</td><td align="left" rowspan="1" colspan="1">0.592</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">0.786</td><td align="left" rowspan="1" colspan="1">
<bold>0.865</bold>
</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">0.844</td></tr><tr><td align="left" rowspan="1" colspan="1"> Precision</td><td align="left" rowspan="1" colspan="1">0.430</td><td align="left" rowspan="1" colspan="1">0.382</td><td align="left" rowspan="1" colspan="1">0.750</td><td align="left" rowspan="1" colspan="1">0.681</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">
<bold>0.767</bold>
</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">0.750</td></tr><tr><td align="left" rowspan="1" colspan="1"> Recall</td><td align="left" rowspan="1" colspan="1">0.353</td><td align="left" rowspan="1" colspan="1">0.187</td><td align="left" rowspan="1" colspan="1">0.378</td><td align="left" rowspan="1" colspan="1">0.230</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">0.408</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">
<bold>0.650</bold>
</td></tr><tr><td align="left" colspan="10" rowspan="1">JL10</td></tr><tr><td align="left" rowspan="1" colspan="1"> MCC</td><td align="left" rowspan="1" colspan="1">0.046</td><td align="left" rowspan="1" colspan="1">0.007</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">0.083</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">0.211</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">
<bold>0.293</bold>
</td></tr><tr><td align="left" rowspan="1" colspan="1"> AUROC</td><td align="left" rowspan="1" colspan="1">0.477</td><td align="left" rowspan="1" colspan="1">0.504</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">0.532</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">0.592</td><td align="left" rowspan="1" colspan="1">0.592</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">
<bold>0.739</bold>
</td></tr><tr><td align="left" rowspan="1" colspan="1"> Precision</td><td align="left" rowspan="1" colspan="1">0.295</td><td align="left" rowspan="1" colspan="1">0.338</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">0.433</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">0.549</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">
<bold>0.582</bold>
</td></tr><tr><td align="left" rowspan="1" colspan="1"> Recall</td><td align="left" rowspan="1" colspan="1">0.194</td><td align="left" rowspan="1" colspan="1">0.131</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">0.142</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">0.296</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">
<bold>0.531</bold>
</td></tr><tr><td align="left" colspan="10" rowspan="1">TL12</td></tr><tr><td align="left" rowspan="1" colspan="1"> MCC</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">0.440</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">
<bold>0.546</bold>
</td></tr><tr><td align="left" rowspan="1" colspan="1"> AUROC</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">0.704</td><td align="left" rowspan="1" colspan="1">0.754</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">
<bold>0.816</bold>
</td></tr><tr><td align="left" rowspan="1" colspan="1"> Precision</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">0.740</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">
<bold>0.816</bold>
</td></tr><tr><td align="left" rowspan="1" colspan="1"> Recall</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">0.514</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">
<bold>0.599</bold>
</td></tr><tr><td align="left" colspan="10" rowspan="1">TE18</td></tr><tr><td align="left" rowspan="1" colspan="1"> MCC</td><td align="left" rowspan="1" colspan="1">0.071</td><td align="left" rowspan="1" colspan="1">0.010</td><td align="left" rowspan="1" colspan="1">0.253</td><td align="left" rowspan="1" colspan="1">0.187</td><td align="left" rowspan="1" colspan="1">0.324</td><td align="left" rowspan="1" colspan="1">0.327</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">
<bold>0.351</bold>
</td><td align="left" rowspan="1" colspan="1">0.190</td></tr><tr><td align="left" rowspan="1" colspan="1"> AUROC</td><td align="left" rowspan="1" colspan="1">0.590</td><td align="left" rowspan="1" colspan="1">0.474</td><td align="left" rowspan="1" colspan="1">0.776</td><td align="left" rowspan="1" colspan="1">0.559</td><td align="left" rowspan="1" colspan="1">0.720</td><td align="left" rowspan="1" colspan="1">0.709</td><td align="left" rowspan="1" colspan="1">0.737</td><td align="left" rowspan="1" colspan="1">
<bold>0.745</bold>
</td><td align="left" rowspan="1" colspan="1">0.647</td></tr><tr><td align="left" rowspan="1" colspan="1"> Precision</td><td align="left" rowspan="1" colspan="1">0.449</td><td align="left" rowspan="1" colspan="1">0.370</td><td align="left" rowspan="1" colspan="1">0.675</td><td align="left" rowspan="1" colspan="1">0.655</td><td align="left" rowspan="1" colspan="1">0.681</td><td align="left" rowspan="1" colspan="1">
<bold>0.729</bold>
</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">0.645</td><td align="left" rowspan="1" colspan="1">0.562</td></tr><tr><td align="left" rowspan="1" colspan="1"> Recall</td><td align="left" rowspan="1" colspan="1">0.288</td><td align="left" rowspan="1" colspan="1">0.214</td><td align="left" rowspan="1" colspan="1">0.263</td><td align="left" rowspan="1" colspan="1">0.173</td><td align="left" rowspan="1" colspan="1">0.345</td><td align="left" rowspan="1" colspan="1">
<bold>0.379</bold>
</td><td align="left" rowspan="1" colspan="1">–</td><td align="left" rowspan="1" colspan="1">0.342</td><td align="left" rowspan="1" colspan="1">0.357</td></tr></tbody></table><table-wrap-foot><fn id="tblfn3"><p>
<sup>a</sup>The values of other binding site prediction methods were taken from Gao <italic toggle="yes">et al.</italic> [<xref rid="ref12" ref-type="bibr">12</xref>]. RNABind and MVRBind results were taken from Zhu <italic toggle="yes">et al.</italic> [<xref rid="ref13" ref-type="bibr">13</xref>] and Chen <italic toggle="yes">et al.</italic> [<xref rid="ref16" ref-type="bibr">16</xref>], respectively. The highest value for each metric is highlighted in bold.</p></fn></table-wrap-foot></table-wrap><p>One of the successful cases of CoBRA is illustrated in <xref rid="f3" ref-type="fig">Fig. 3A</xref>. A structure of pir-miRNA-300 apical loop fused to ydaO riboswitch scaffold, complexed with c-di-AMP (PDB ID: 6WTR) from the JL10 dataset. The switch regulates the gene expression in bacteria by sensing concentrations of ATP and c-di-AMP. The RNA structure is comprised of two three-way junctions connected by two helices and a large conserved interior loop. The compound works as an off switch when binding to the RNA. The binding site of the molecule is located between the two pseudoknots, stabilizing the RNA structure through intermolecular stacking [<xref rid="ref33" ref-type="bibr">33</xref>, <xref rid="ref34" ref-type="bibr">34</xref>]. A comparison of the two programs revealed that CoBRA exhibited superior performance in comparison with ZeSTa. The MCC, precision, and recall of CoBRA are 0.923, 0.946, and 0.946, respectively, while those of ZeSTa are 0.121, 0.455, and 0.142, respectively. The uniqueness of the binding site characteristics makes the structure-based binding site prediction methods hard to predict, while CoBRA, a sequence-based one, demonstrated superior performance.</p><fig position="float" id="f3" orientation="portrait"><label>Figure 3</label><caption><p>Case studies of CoBRA. (A) A successfully predicted case of CoBRA. A pir-miRNA-300 complexed with bis-(3′,5′)-cyclic-dimeric-adenosine-monophosphate (PDB ID: 6WTR). (B)RNA Aptamer complexed with flavin mononucleotide (PDB ID: 1FMN). The right panel displays the distribution of Laplacian norm values, where blue indicates lower curvature (concave) and red indicates higher curvature (convex). (C) Wrongly predicted case of CoBRA: sisomicin bound to bacterial ribosomal decoding site (PDB ID: 4F8U). The right panel shows an RNA homodimer, highlighting the inter-chain binding interface.</p></caption><graphic xmlns:xlink="http://www.w3.org/1999/xlink" position="float" orientation="portrait" xlink:href="bbaf713f3.jpg"><?image-name bbaf713f3.jpg?><?image-size 24991?><?image-md5 10343df9ea0ed4acd095b1831e48db93?><?image-image-server-status LOAD_COMPLETED?><?image-original-height 181?><?image-original-width 1133?><?image-scaled-height 121?><?image-scaled-width 755?><?image-cloudpmc-urn urn:cdn:blobs/9241/12790621/10343df9ea0e/bbaf713f3.jpg?><?thumb-name bbaf713f3.gif?><?thumb-size 3837?><?thumb-md5 079d8e20d984f291a35cfd9977a2463e?><?thumb-image-server-status NEVER_LOAD?><?thumb-scaled-height 32?><?thumb-scaled-width 200?><?thumb-cloudpmc-urn urn:cdn:blobs/9241/12790621/079d8e20d984/bbaf713f3.gif?><alt-text>Alt Text: Structural examples illustrate good and bad binding site predictions made by CoBRA.</alt-text></graphic></fig><p>The computational cost of CoBRA was then compared with that of the LR and RF models. The generation of the embedding using RiNALMo required a time period of 111 ms per sequence. Utilizing fixed embeddings, the MLP required 109.97 s for training and 0.222 s for testing, while the LR necessitated 44.34 s and 0.008 s, and the RF employed 74.79 s and 0.101 s.</p></sec><sec id="sec17"><title>Structural analysis of TE18 dataset</title><p>An investigation was conducted to examine the outcomes of TE18, wherein CoBRA exhibited the least optimal performance among the test sets. Dissecting the spatial structure of RNA (DSSR) [<xref rid="ref35" ref-type="bibr">35</xref>] was employed to analyze the secondary structures of RNA crystal structures. The RNA secondary structures were classified into eight categories: stem, canonical, isolated canonical, internal loop, hairpin loop, bulge, junction, and helix. A subsequent analysis of the prediction performance of CoBRA was conducted, with the structural type serving as the primary variable.</p><p>CoBRA demonstrated a consistent level of accuracy, with a minimum of 51% for identifying binding sites across structural types. The highest accuracy was observed in junction sites, with an accuracy of 92%. On the other hand, the performance for identifying non-binding nucleotides exhibited variability according to structure, with higher accuracy in the stem (68%) and junction (67%), but lower accuracy in internal loops (47%) (<xref rid="sup1" ref-type="supplementary-material">Supplementary Fig. S1</xref>).</p><p>We also conducted a residue-level curvature analysis of the TE18 dataset by calculating the LN. As the LN values increase, the geometry of the residue evolves from concave to convex. The LN values of the binding and non-binding sites show distinct distribution (<xref rid="sup1" ref-type="supplementary-material">Supplementary Fig. S2</xref>), suggesting structural differences between the two classes. The binding site nucleotides have LN values ranging from 1.5 to 20.7, with an average of 10.9. Conversely, the non-binding ones exhibit larger values, ranging from 2.3 to 23.0, with an average of 12.7. It can be inferred that the geometry of RNA compound binding sites is relatively concave, a finding that is also reported by Su <italic toggle="yes">et al.</italic> [<xref rid="ref11" ref-type="bibr">11</xref>].</p><p>Despite the absence of explicit incorporation of structural characteristics such as concavity as an input feature in CoBRA, a statistically significant relationship was observed between the model predictions and the LN value, as indicated by a point-biserial correlation of −0.171 with a <italic toggle="yes">P</italic>-value of 2.8e-5. This demonstrates that a model trained solely on RNA sequence embeddings could explain structural characteristics via inherent sequence patterns. We also observed a correlation of −0.247 between LN value and the actual binding label, indicating that actual binding sites tend to be located in structurally concave regions. Correlation analyses with TP/TN/FP/FN from CoBRA prediction demonstrate that TP and TN exhibit relatively high correlation coefficients (TP: r = −0.188, <italic toggle="yes">P</italic> = 4.0e-6; TN: r = 0.260, <italic toggle="yes">P</italic> = 1.2e-10). No significant relationship was found for FP (r = −0.027, <italic toggle="yes">P</italic> = 0.505), and FN exhibited a modest negative correlation (r = −0.126, <italic toggle="yes">P</italic> = 2.0e-3), indicating a slight reduction in missed positive predictions. A notable example is provided in <xref rid="f3" ref-type="fig">Fig. 3B</xref> (PDB ID: 1FMN), illustrating the substantial correlation between CoBRA prediction and LN values. The flavin binding site of an RNA aptamer, defined by 4 Å from the co-crystalized ligand, exhibited a − 0.541 correlation coefficient (<italic toggle="yes">P</italic> = 8.0e-4). The right panel displays the LN values of nucleotides, indicated by the color change from blue (concave) to red (convex). CoBRA demonstrates an accuracy in predicting the binding site, with a precision of 0.636. However, the program failed to accurately predict three nucleotides with low LN values (average: 8.36) and erroneously predicted four nucleotides with high LN values (average: 9.89). This finding might show a limitation of only using RNA sequence information.</p><p>One of the most severe predicted cases of CoBRA in the TE18 dataset is sisomicin complexed with the bacterial ribosomal decoding site (PDB ID: 4F8U, <xref rid="f3" ref-type="fig">Fig. 3C</xref>). The crystal structure contains homodimer RNA chains, and two sisomicins bind to the interface of the dimer. For prediction, the RNA sequence of a single chain, B chain of the PDB structure, was used as the input. CoBRA demonstrated a successful prediction for one of the binding sites (the red-colored region of the left panel). However, given that the binding site is designated by the ligand with the same chain ID of the input RNA sequence (the blue region), the resulting accuracy for that particular sample was found to be 0.0. A visual inspection of both the A and B chains together (right panel) revealed that the region was misclassified as a false positive. This case demonstrates that the accuracy of prediction may be diminished when binding sites are situated at the interface with other chains of a query sequence. This phenomenon has also been observed in protein-ligand binding site prediction problems using LMs [<xref rid="ref36" ref-type="bibr">36</xref>]. One potential solution to this issue involves the inference of structural information or the incorporation of multi-chain as an input.</p></sec><sec id="sec18"><title>Prediction results on metal binding sites</title><p>Metal ions such as Mg<sup>2+</sup> and Na<sup>+</sup> often bind diffusely across the RNA surface to neutralize the negatively charged phosphate backbone, without forming specific binding pockets. These ions contribute to RNA structural stability and folding, as previously noted by Draper <italic toggle="yes">et al.</italic> [<xref rid="ref37" ref-type="bibr">37</xref>]. In contrast, organic molecules typically bind within well-defined pockets, making their spatial localization and prediction of binding sites more tractable. Thus, predicting metal binding sites may be more challenging than predicting organic molecule binding sites due to the nature of metal binding. Also, due to the electrostatic and steric nature of metal ion binding, they may associate with multiple structurally similar regions with comparable affinity, complicating the prediction task, which is reported in MetalionRNA [<xref rid="ref38" ref-type="bibr">38</xref>].</p><p>The binding sites of the test sets were classified according to the type of binding molecule: metal ion or non-metallic compound. The accuracies of metal binding site prediction were 0.647, 0.531, 0.497, and 0.202 for RB9, JL10, TL12, and TE18, respectively. As anticipated, the non-metallic compound category demonstrated higher levels of accuracy, with the values obtained for RB9, JL10, TL12, and TE18 being 0.733, 0.739, 0.690, and 0.468, respectively. Detailed information is summarized in <xref rid="sup1" ref-type="supplementary-material">Supplementary Table S6</xref>. This discrepancy is likely attributable to the inherent nature of RNA–metal ion interactions. The incorporation of physico-chemical characteristics has the potential to enhance the efficacy of predicting metal binding sites.</p></sec></sec><sec id="sec19"><title>Discussion</title><p>Given the recent emphasis on RNA as a promising therapeutic target, the prediction of compound binding sites could serve as a fundamental starting point. The RNA-ligand binding site prediction models can be broadly categorized into two distinct approaches: structure-based and sequence-based. The structure-based approach leverages three-dimensional or secondary structural information to capture critical spatial and geometric features, while the sequence-based approach utilizes evolutionary and contextual patterns through sequence embeddings. While structure-based models generally demonstrate robust performance when accurate structural information is available, they exhibit significant performance decrement on complex RNA architectures, such as junction loops. Consistent with this observation, CoBRA showed lower AUROC values than structure-aware models on the RNABind dataset, reflecting the limitation of sequence-only representations in capturing structural determinants of ligand binding.</p><p>The integration of ERNIE-RNA as an RNA LM and focal loss function has led to the development of a novel RNA-compound binding site program, called CoBRA. On various benchmark sets, the model demonstrated performance that was either superior to or comparable to contemporary state-of-the-art prediction methods. The program demonstrated its capacity for generalization across a range of external datasets, including RB9 and TL12, without requiring explicit structural inputs. This finding highlights the efficacy of sequence embeddings in capturing functional signals across a diverse array of RNA architectures.</p><p>Despite these advances, several limitations were identified through structural analysis on the TE18 set, in which CoBRA exhibited suboptimal performance. The performance of the program is contingent upon the RNA secondary structure, denoted by DSSR. Also, it misses predicting binding site nucleotides located in concave regions, as indicated by low LN values. In addition, binding sites located at interfaces of RNA dimers were often missed when only a single RNA chain was provided as input. Although the program demonstrated comparable performance to the state-of-the-art methods, only using LMs, the absence of structural information could lead to failure in predicting the binding site. Consequently, the subsequent direction for CoBRA will be to address these deficiencies.</p><boxed-text id="box01" position="float" orientation="portrait"><sec id="sec3"><title>Key Points</title><list list-type="bullet"><list-item><p>Compound binding site prediction for RNA (CoBRA) provides a lightweight deep-learning method for RNA-ligand binding site prediction.</p></list-item><list-item><p>A comprehensive benchmark combining RNA language models and loss functions to find the optimal combination for CoBRA.</p></list-item><list-item><p>CoBRA can effectively predict RNA-ligand binding sites with a superior performance to the other state-of-the-art methods.</p></list-item></list></sec></boxed-text></sec><sec sec-type="supplementary-material"><title>Supplementary Material</title><supplementary-material id="sup1" position="float" content-type="local-data" orientation="portrait"><label>SI_rev_submit_bbaf713</label><media xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="si_rev_submit_bbaf713.docx" position="float" orientation="portrait"><?suppdata-name si_rev_submit_bbaf713.docx?><?suppdata-size 212522?><?suppdata-md5 dc498934a2adb7943ae21dfd39791c42?><?suppdata-image-server-status NEVER_LOAD?><?suppdata-mime-type application?><?suppdata-mime-sub-type vnd.openxmlformats-officedocument.wordprocessingml.document?><?suppdata-cloudpmc-urn urn:app:9241/12790621/dc498934a2ad/si_rev_submit_bbaf713.docx?></media></supplementary-material></sec></body><back><sec id="sec21"><title>Author contributions</title><p>Woong-Hee Shin conceived the study. Wonkyeong Jang designed and implemented CoBRA and conducted the benchmark. Woong-Hee Shin and Wonkyeong Jang analyzed the results. Wonkyeong Jang composed the manuscript. Woong-Hee Shin revised and polished the article. All authors have read and approved the final version of the manuscript.</p><p>Conflict of interest: None declared.</p></sec><sec id="sec23"><title>Funding</title><p>This work was supported by the Institute of Information &amp; Communications Technology Planning &amp; Evaluation (IITP)—ICT Challenge and Advanced Network of HRD (ICAN) grant funded by the Korea government (Ministry of Science and ICT) [IITP-2025-RS-2022-00156439 and IITP-2025-RS-2024-00438263]; the Bio &amp; Medical Technology Development Program of the National Research Foundation (NRF) funded by the Korea government (Ministry of Science and ICT) [RS-2025-02217289 and 2022M3E5F3081268 to WHS]; Korea University grant [K2517281 to WHS].</p></sec><sec sec-type="data-availability" id="sec24"><title>Data availability</title><p>All datasets and source code are available at <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://github.com/kucm-lsbi/CoBRA" ext-link-type="uri">https://github.com/kucm-lsbi/CoBRA</ext-link>.</p></sec><ref-list id="bib1"><title>References</title><ref id="ref1"><label>1.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author">
<string-name name-style="western">
<surname>Esteller</surname>  <given-names>M</given-names></string-name>
</person-group>. <article-title>Non-coding RNAs in human disease</article-title>. <source><italic toggle="yes">Nat Rev Genet</italic></source>  <year>2011</year>;<volume>12</volume>:<fpage>861</fpage>–<lpage>74</lpage>. <pub-id pub-id-type="doi">10.1038/nrg3074</pub-id><pub-id pub-id-type="pmid">22094949</pub-id>
</mixed-citation></ref><ref id="ref2"><label>2.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author">
<string-name name-style="western">
<surname>Chen</surname>  <given-names>G</given-names></string-name>, <string-name name-style="western"><surname>Wang</surname>  <given-names>Z</given-names></string-name>, <string-name name-style="western"><surname>Wang</surname>  <given-names>D</given-names></string-name>. <etal>et al.</etal></person-group>  <article-title>LncRNADisease: a database for long-non-coding RNA-associated diseases</article-title>. <source><italic toggle="yes">Nucleic Acids Res</italic></source>  <year>2013</year>;<volume>41</volume>:<fpage>D983</fpage>–<lpage>6</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gks1099</pub-id><pub-id pub-id-type="pmid">23175614</pub-id>
<pub-id pub-id-type="pmcid">PMC3531173</pub-id></mixed-citation></ref><ref id="ref3"><label>3.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author">
<string-name name-style="western">
<surname>Lu</surname>  <given-names>M</given-names></string-name>, <string-name name-style="western"><surname>Zhang</surname>  <given-names>Q</given-names></string-name>, <string-name name-style="western"><surname>Deng</surname>  <given-names>M</given-names></string-name>. <etal>et al.</etal></person-group>  <article-title>An analysis of human microRNA and disease associations</article-title>. <source><italic toggle="yes">PloS One</italic></source>  <year>2008</year>;<volume>3</volume>:<fpage>e3420</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0003420</pub-id><pub-id pub-id-type="pmid">18923704</pub-id>
<pub-id pub-id-type="pmcid">PMC2559869</pub-id></mixed-citation></ref><ref id="ref4"><label>4.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author">
<string-name name-style="western">
<prefix>de</prefix>  <surname>Souza</surname>  <given-names>N</given-names></string-name>
</person-group>. <article-title>The ENCODE project</article-title>. <source><italic toggle="yes">Nat Methods</italic></source>  <year>2012</year>;<volume>10</volume>:<fpage>1046</fpage>. <pub-id pub-id-type="doi">10.1038/nmeth.2238</pub-id><pub-id pub-id-type="pmid">23281567</pub-id></mixed-citation></ref><ref id="ref5"><label>5.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author">
<string-name name-style="western">
<surname>Hopkins</surname>  <given-names>A</given-names></string-name>, <string-name name-style="western"><surname>Groom</surname>  <given-names>C</given-names></string-name></person-group>. <article-title>The druggable genome</article-title>. <source><italic toggle="yes">Nat Rev Drug Discov</italic></source>  <year>2002</year>;<volume>1</volume>:<fpage>727</fpage>–<lpage>30</lpage>. <pub-id pub-id-type="doi">10.1038/nrd892</pub-id><pub-id pub-id-type="pmid">12209152</pub-id>
</mixed-citation></ref><ref id="ref6"><label>6.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author">
<string-name name-style="western">
<surname>Shao</surname>  <given-names>Y</given-names></string-name>, <string-name name-style="western"><surname>Zhang</surname>  <given-names>QC</given-names></string-name></person-group>. <article-title>Targeting RNA structures in diseases with small molecules</article-title>. <source><italic toggle="yes">Essays Biochem</italic></source>  <year>2020</year>;<volume>64</volume>:<fpage>955</fpage>–<lpage>66</lpage>. <pub-id pub-id-type="doi">10.1042/EBC20200011</pub-id><pub-id pub-id-type="pmid">33078198</pub-id>
<pub-id pub-id-type="pmcid">PMC7724634</pub-id></mixed-citation></ref><ref id="ref7"><label>7.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author">
<string-name name-style="western">
<surname>Yu</surname>  <given-names>A-M</given-names></string-name>, <string-name name-style="western"><surname>Choi</surname>  <given-names>YH</given-names></string-name>, <string-name name-style="western"><surname>Tu</surname>  <given-names>M-J</given-names></string-name></person-group>. <article-title>RNA drugs and RNA targets for small molecules: principles, progress, and challenges</article-title>. <source><italic toggle="yes">Pharmacol Rev</italic></source>  <year>2020</year>;<volume>72</volume>:<fpage>862</fpage>–<lpage>98</lpage>. <pub-id pub-id-type="doi">10.1124/pr.120.019554</pub-id><pub-id pub-id-type="pmid">32929000</pub-id>
<pub-id pub-id-type="pmcid">PMC7495341</pub-id></mixed-citation></ref><ref id="ref8"><label>8.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author">
<string-name name-style="western">
<surname>Zeng</surname>  <given-names>P</given-names></string-name>, <string-name name-style="western"><surname>Li</surname>  <given-names>J</given-names></string-name>, <string-name name-style="western"><surname>Ma</surname>  <given-names>W</given-names></string-name>. <etal>et al.</etal></person-group>  <article-title>Rsite: a computational method to identify the functional sites of noncoding RNAs</article-title>. <source><italic toggle="yes">Sci Rep</italic></source>  <year>2015</year>;<volume>5</volume>:<fpage>9179</fpage>. <pub-id pub-id-type="doi">10.1038/srep09179</pub-id><pub-id pub-id-type="pmid">25776805</pub-id>
<pub-id pub-id-type="pmcid">PMC4361870</pub-id></mixed-citation></ref><ref id="ref9"><label>9.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author">
<string-name name-style="western">
<surname>Zeng</surname>  <given-names>P</given-names></string-name>, <string-name name-style="western"><surname>Cui</surname>  <given-names>Q</given-names></string-name></person-group>. <article-title>Rsite2: an efficient computational method to predict the functional sites of noncoding RNAs</article-title>. <source><italic toggle="yes">Sci Rep</italic></source>  <year>2016</year>;<volume>6</volume>:<fpage>19016</fpage>. <pub-id pub-id-type="doi">10.1038/srep19016</pub-id><pub-id pub-id-type="pmid">26751501</pub-id>
<pub-id pub-id-type="pmcid">PMC4707467</pub-id></mixed-citation></ref><ref id="ref10"><label>10.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author">
<string-name name-style="western">
<surname>Wang</surname>  <given-names>K</given-names></string-name>, <string-name name-style="western"><surname>Jian</surname>  <given-names>Y</given-names></string-name>, <string-name name-style="western"><surname>Wang</surname>  <given-names>H</given-names></string-name>. <etal>et al.</etal></person-group>  <article-title>RBind: computational network method to predict RNA binding sites</article-title>. <source><italic toggle="yes">Bioinformatics</italic></source>  <year>2018</year>;<volume>34</volume>:<fpage>3131</fpage>–<lpage>6</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty345</pub-id><pub-id pub-id-type="pmid">29718097</pub-id>
</mixed-citation></ref><ref id="ref11"><label>11.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author">
<string-name name-style="western">
<surname>Su</surname>  <given-names>H</given-names></string-name>, <string-name name-style="western"><surname>Peng</surname>  <given-names>Z</given-names></string-name>, <string-name name-style="western"><surname>Yang</surname>  <given-names>J</given-names></string-name></person-group>. <article-title>Recognition of small molecule–RNA binding sites using RNA sequence and structure</article-title>. <source><italic toggle="yes">Bioinformatics</italic></source>  <year>2021</year>;<volume>37</volume>:<fpage>36</fpage>–<lpage>42</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btaa1092</pub-id><pub-id pub-id-type="pmid">33416863</pub-id>
<pub-id pub-id-type="pmcid">PMC8034527</pub-id></mixed-citation></ref><ref id="ref12"><label>12.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author">
<string-name name-style="western">
<surname>Gao</surname>  <given-names>J</given-names></string-name>, <string-name name-style="western"><surname>Liu</surname>  <given-names>H</given-names></string-name>, <string-name name-style="western"><surname>Zhuo</surname>  <given-names>C</given-names></string-name>. <etal>et al.</etal></person-group>  <article-title>Predicting small molecule binding nucleotides in RNA structures using RNA surface topography</article-title>. <source><italic toggle="yes">J Chem Inf Model</italic></source>  <year>2024</year>;<volume>64</volume>:<fpage>6979</fpage>–<lpage>92</lpage>. <pub-id pub-id-type="doi">10.1092/acs.jcim.4c01264</pub-id><pub-id pub-id-type="pmid">39230508</pub-id>
</mixed-citation></ref><ref id="ref13"><label>13.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author">
<string-name name-style="western">
<surname>Zhu</surname>  <given-names>W</given-names></string-name>, <string-name name-style="western"><surname>Ding</surname>  <given-names>X</given-names></string-name>, <string-name name-style="western"><surname>Shen</surname>  <given-names>H</given-names></string-name>. <etal>et al.</etal></person-group>  <article-title>Identifying RNA-small molecule binding sites using geometric deep learning with language models</article-title>. <source><italic toggle="yes">J Mol Biol</italic></source>  <year>2025</year>;<volume>437</volume>. <pub-id pub-id-type="doi">10.1016/j.jmb.2025.169010</pub-id><pub-id pub-id-type="pmid">39961524</pub-id></mixed-citation></ref><ref id="ref14"><label>14.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author">
<string-name name-style="western">
<surname>Sun</surname>  <given-names>S</given-names></string-name>, <string-name name-style="western"><surname>Yang</surname>  <given-names>J</given-names></string-name>, <string-name name-style="western"><surname>Gao</surname>  <given-names>L</given-names></string-name>. <etal>et al.</etal></person-group>  <article-title>RNA language model and graph attention network for RNA and small molecule binding sites prediction</article-title>. <source><italic toggle="yes">Bioinformatics</italic></source>  <year>2025</year>;<volume>41</volume>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btaf447</pub-id><pub-id pub-id-type="pmcid">PMC12417085</pub-id><pub-id pub-id-type="pmid">40795032</pub-id></mixed-citation></ref><ref id="ref15"><label>15.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author">
<string-name name-style="western">
<surname>Sun</surname>  <given-names>C</given-names></string-name>, <string-name name-style="western"><surname>Zhang</surname>  <given-names>L</given-names></string-name>, <string-name name-style="western"><surname>Zhang</surname>  <given-names>L</given-names></string-name>. <etal>et al.</etal></person-group>  <article-title>GATRsite: RNA–ligand binding site prediction using graph attention networks and pretrained RNA language models</article-title>. <source><italic toggle="yes">J Chem Inf Model</italic></source>  <year>2025</year>;<volume>65</volume>:<fpage>8448</fpage>–<lpage>61</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.5c00605</pub-id><pub-id pub-id-type="pmid">40814145</pub-id>
</mixed-citation></ref><ref id="ref16"><label>16.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author">
<string-name name-style="western">
<surname>Chen</surname>  <given-names>S</given-names></string-name>, <string-name name-style="western"><surname>Huang</surname>  <given-names>Z</given-names></string-name>, <string-name name-style="western"><surname>Wang</surname>  <given-names>Y</given-names></string-name>. <etal>et al.</etal></person-group>  <article-title>MVRBind: multi-view learning for RNA-small molecule binding site prediction</article-title>. <source><italic toggle="yes">Brief Bioinform</italic></source>  <year>2025</year>;<volume>26</volume>:<fpage>bbaf489</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbaf489</pub-id><pub-id pub-id-type="pmid">40977268</pub-id>
<pub-id pub-id-type="pmcid">PMC12451103</pub-id></mixed-citation></ref><ref id="ref17"><label>17.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author">
<string-name name-style="western">
<surname>Panei</surname>  <given-names>FP</given-names></string-name>, <string-name name-style="western"><surname>Torchet</surname>  <given-names>R</given-names></string-name>, <string-name name-style="western"><surname>Menager</surname>  <given-names>H</given-names></string-name>. <etal>et al.</etal></person-group>  <article-title>HARIBOSS: a curated database of RNA-small molecules structures to aid rational drug design</article-title>. <source><italic toggle="yes">Bioinformatics</italic></source>  <year>2022</year>;<volume>38</volume>:<fpage>4185</fpage>–<lpage>93</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btac483</pub-id><pub-id pub-id-type="pmid">35799352</pub-id>
</mixed-citation></ref><ref id="ref18"><label>18.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author">
<string-name name-style="western">
<surname>Paszke</surname>  <given-names>A</given-names></string-name>
</person-group>. <article-title>Pytorch: an imperative style, high-performance deep learning library</article-title>. <source><italic toggle="yes">arXiv</italic></source>  <year>2019</year>. <pub-id pub-id-type="doi">10.48550/arXiv.1912.01703</pub-id></mixed-citation></ref><ref id="ref19"><label>19.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author">
<string-name name-style="western">
<surname>Loshchilov</surname>  <given-names>I</given-names></string-name>, <string-name name-style="western"><surname>Hutter</surname>  <given-names>F</given-names></string-name></person-group>. <article-title>Decoupled weight decay regularization</article-title>. <source><italic toggle="yes">arXiv</italic></source>  <year>2019</year>. <pub-id pub-id-type="doi">10.48550/arXiv.1711.05101</pub-id></mixed-citation></ref><ref id="ref20"><label>20.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author">
<string-name name-style="western">
<surname>Yin</surname>  <given-names>W</given-names></string-name>, <string-name name-style="western"><surname>Zhang</surname>  <given-names>Z</given-names></string-name>, <string-name name-style="western"><surname>He</surname>  <given-names>L</given-names></string-name>. <etal>et al.</etal></person-group>  <article-title>ERNIE-RNA: an RNA language model with structure-enhanced representations</article-title>. <source><italic toggle="yes">bioRxiv</italic></source>  <year>2024</year>. <pub-id pub-id-type="doi">10.1101/2024.03.17.585376</pub-id><pub-id pub-id-type="pmcid">PMC12627772</pub-id><pub-id pub-id-type="pmid">41253752</pub-id></mixed-citation></ref><ref id="ref21"><label>21.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author">
<string-name name-style="western">
<surname>Penić</surname>  <given-names>RJ</given-names></string-name>, <string-name name-style="western"><surname>Vlašić</surname>  <given-names>T</given-names></string-name>, <string-name name-style="western"><surname>Huber</surname>  <given-names>RG</given-names></string-name>. <etal>et al.</etal></person-group>  <article-title>RiNALMo: general-purpose RNA language models can generalize well on structure prediction tasks</article-title>. <source><italic toggle="yes">Nat Commun</italic></source>  <year>2025</year>;<volume>16</volume>:<fpage>5671</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-025-60872-5</pub-id><pub-id pub-id-type="pmid">40593636</pub-id>
<pub-id pub-id-type="pmcid">PMC12219582</pub-id></mixed-citation></ref><ref id="ref22"><label>22.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author">
<string-name name-style="western">
<surname>Akiyama</surname>  <given-names>M</given-names></string-name>, <string-name name-style="western"><surname>Sakakibara</surname>  <given-names>Y</given-names></string-name></person-group>. <article-title>Informative RNA base embedding for RNA structural alignment and clustering by deep representation learning</article-title>. <source><italic toggle="yes">NAR Genomics Bioinf</italic></source>  <year>2022</year>;<volume>4</volume>:<fpage>lqac012</fpage>. <pub-id pub-id-type="doi">10.1093/nargab/lqac012</pub-id><pub-id pub-id-type="pmcid">PMC8862729</pub-id><pub-id pub-id-type="pmid">35211670</pub-id></mixed-citation></ref><ref id="ref23"><label>23.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author">
<string-name name-style="western">
<surname>Chen</surname>  <given-names>J</given-names></string-name>, <string-name name-style="western"><surname>Hu</surname>  <given-names>Z</given-names></string-name>, <string-name name-style="western"><surname>Sun</surname>  <given-names>S</given-names></string-name>. <etal>et al.</etal></person-group>  <article-title>Interpretable RNA foundation model from unannotated data for highly accurate RNA structure and function predictions</article-title>. <source><italic toggle="yes">arXiv</italic></source>  <year>2022</year>. <pub-id pub-id-type="doi">10.48550/arXiv.2204.00300</pub-id></mixed-citation></ref><ref id="ref24"><label>24.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author">
<string-name name-style="western">
<surname>Zhang</surname>  <given-names>Y</given-names></string-name>, <string-name name-style="western"><surname>Lang</surname>  <given-names>M</given-names></string-name>, <string-name name-style="western"><surname>Jiang</surname>  <given-names>J</given-names></string-name>. <etal>et al.</etal></person-group>  <article-title>Multiple sequence alignment-based RNA language model and its application to structural inference</article-title>. <source><italic toggle="yes">Nucleic Acids Res</italic></source>  <year>2024</year>;<volume>52</volume>:<fpage>e3</fpage>. <pub-id pub-id-type="doi">10.1093/nar/gkad1031</pub-id><pub-id pub-id-type="pmid">37941140</pub-id>
<pub-id pub-id-type="pmcid">PMC10783488</pub-id></mixed-citation></ref><ref id="ref25"><label>25.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author">
<string-name name-style="western">
<surname>Chen</surname>  <given-names>K</given-names></string-name>, <string-name name-style="western"><surname>Zhou</surname>  <given-names>Y</given-names></string-name>, <string-name name-style="western"><surname>Ding</surname>  <given-names>M</given-names></string-name>. <etal>et al.</etal></person-group>  <article-title>Self-supervised learning on millions of primary RNA sequences from 72 vertebrates improves sequence-based RNA splicing prediction</article-title>. <source><italic toggle="yes">Brief Bioinform</italic></source>  <year>2024</year>;<volume>25</volume>:<fpage>bbae163</fpage>. <pub-id pub-id-type="doi">10.1093/bib/bbae163</pub-id><pub-id pub-id-type="pmid">38605640</pub-id>
<pub-id pub-id-type="pmcid">PMC11009468</pub-id></mixed-citation></ref><ref id="ref26"><label>26.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author">
<string-name name-style="western">
<surname>Chu</surname>  <given-names>Y</given-names></string-name>, <string-name name-style="western"><surname>Yu</surname>  <given-names>D</given-names></string-name>, <string-name name-style="western"><surname>Li</surname>  <given-names>Y</given-names></string-name>. <etal>et al.</etal></person-group>  <article-title>A 5′ UTR language model for decoding untranslated regions of mRNA and function predictions</article-title>. <source><italic toggle="yes">Nat Mach Intell</italic></source>  <year>2024</year>;<volume>6</volume>:<fpage>449</fpage>–<lpage>60</lpage>. <pub-id pub-id-type="doi">10.1038/s42256-024-00823-9</pub-id><pub-id pub-id-type="pmid">38855263</pub-id>
<pub-id pub-id-type="pmcid">PMC11155392</pub-id></mixed-citation></ref><ref id="ref27"><label>27.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author">
<string-name name-style="western">
<surname>Cui</surname>  <given-names>Y</given-names></string-name>, <string-name name-style="western"><surname>Jia</surname>  <given-names>M</given-names></string-name>, <string-name name-style="western"><surname>Lin</surname>  <given-names>T-Y</given-names></string-name>. <etal>et al.</etal></person-group>  <article-title>Class-balanced loss based on effective number of samples</article-title>. <source><italic toggle="yes">arXiv</italic></source>  <year>2019</year>. <pub-id pub-id-type="doi">10.48550/arXiv.1901.05555</pub-id></mixed-citation></ref><ref id="ref28"><label>28.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author">
<string-name name-style="western">
<surname>Salehi</surname>  <given-names>SSM</given-names></string-name>, <string-name name-style="western"><surname>Erdogmus</surname>  <given-names>D</given-names></string-name>, <string-name name-style="western"><surname>Gholipour</surname>  <given-names>A</given-names></string-name></person-group>. <article-title>Tversky loss function for image segmentation using 3D fully convolutional deep networks</article-title>. <source><italic toggle="yes">arXiv</italic></source>  <year>2017</year>. <pub-id pub-id-type="doi">10.48550/arXiv.1706.05721</pub-id></mixed-citation></ref><ref id="ref29"><label>29.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author">
<string-name name-style="western">
<surname>Milletari</surname>  <given-names>F</given-names></string-name>, <string-name name-style="western"><surname>Navab</surname>  <given-names>N</given-names></string-name>, <string-name name-style="western"><surname>Ahmadi</surname>  <given-names>S-A</given-names></string-name></person-group>. <article-title>V-net: fully convolutional neural networks for volumetric medical image segmentation</article-title>. <source><italic toggle="yes">arXiv</italic></source>  <year>2016</year>. <pub-id pub-id-type="doi">10.48550/arXiv.1606.04797</pub-id></mixed-citation></ref><ref id="ref30"><label>30.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author">
<string-name name-style="western">
<surname>Berman</surname>  <given-names>M</given-names></string-name>, <string-name name-style="western"><surname>Triki</surname>  <given-names>AR</given-names></string-name>, <string-name name-style="western"><surname>Blaschko</surname>  <given-names>MB</given-names></string-name></person-group>. <article-title>The lovász-softmax loss: a tractable surrogate for the optimization of the intersection-over-union measure in neural networks</article-title>. <source><italic toggle="yes">arXiv</italic></source>  <year>2018</year>. <pub-id pub-id-type="doi">10.48550/arXiv.1705.08790</pub-id></mixed-citation></ref><ref id="ref31"><label>31.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author">
<string-name name-style="western">
<surname>He</surname>  <given-names>X</given-names></string-name>, <string-name name-style="western"><surname>Zhou</surname>  <given-names>Y</given-names></string-name>, <string-name name-style="western"><surname>Zhou</surname>  <given-names>Z</given-names></string-name>. <etal>et al.</etal></person-group>  <article-title>Triplet-center loss for multi-view 3d object retrieval</article-title>. <source><italic toggle="yes">arXiv</italic></source>  <year>2018</year>. <pub-id pub-id-type="doi">10.48550/arXiv.1803.06189</pub-id></mixed-citation></ref><ref id="ref32"><label>32.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author">
<string-name name-style="western">
<surname>Wang</surname>  <given-names>J</given-names></string-name>, <string-name name-style="western"><surname>Liu</surname>  <given-names>Y</given-names></string-name>, <string-name name-style="western"><surname>Tian</surname>  <given-names>B</given-names></string-name></person-group>. <article-title>Protein-small molecule binding site prediction based on a pre-trained protein language model with contrastive learning</article-title>. <source><italic toggle="yes">J Chem</italic></source>  <year>2024</year>;<volume>16</volume>:<fpage>125</fpage>. <pub-id pub-id-type="doi">10.1186/s13321-024-00920-2</pub-id><pub-id pub-id-type="pmcid">PMC11542454</pub-id><pub-id pub-id-type="pmid">39506806</pub-id></mixed-citation></ref><ref id="ref33"><label>33.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author">
<string-name name-style="western">
<surname>Shoffner</surname>  <given-names>G</given-names></string-name>, <string-name name-style="western"><surname>Peng</surname>  <given-names>Z</given-names></string-name>, <string-name name-style="western"><surname>Guo</surname>  <given-names>F</given-names></string-name></person-group>. <article-title>Structures of microRNA-precursor apical junctions and loops reveal non-canonical base pairs important for processing</article-title>. <source><italic toggle="yes">bioRxiv</italic></source>  <year>2020</year>. <pub-id pub-id-type="doi">10.1101/2020.05.05.078014</pub-id></mixed-citation></ref><ref id="ref34"><label>34.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author">
<string-name name-style="western">
<surname>Gao</surname>  <given-names>A</given-names></string-name>, <string-name name-style="western"><surname>Serganov</surname>  <given-names>A</given-names></string-name></person-group>. <article-title>Structural insights into recognition of c-di-AMP by by the ydaO riboswitch</article-title>. <source><italic toggle="yes">Nat Chem Biol</italic></source>  <year>2014</year>;<volume>10</volume>:787–92. <pub-id pub-id-type="doi">10.1038/nchembio.1607</pub-id><pub-id pub-id-type="pmcid">PMC4294798</pub-id><pub-id pub-id-type="pmid">25086507</pub-id></mixed-citation></ref><ref id="ref35"><label>35.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author">
<string-name name-style="western">
<surname>Lu</surname>  <given-names>X-J</given-names></string-name>
</person-group>. <article-title>DSSR-enabled innovative schematics of 3D nucleic acid structures with PyMOL</article-title>. <source><italic toggle="yes">Nucleic Acids Res</italic></source>  <year>2020</year>;<volume>48</volume>:<fpage>e74</fpage>. <pub-id pub-id-type="doi">10.1093/nar/gkaa426</pub-id><pub-id pub-id-type="pmid">32442277</pub-id>
<pub-id pub-id-type="pmcid">PMC7367123</pub-id></mixed-citation></ref><ref id="ref36"><label>36.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author">
<string-name name-style="western">
<surname>Chelur</surname>  <given-names>VR</given-names></string-name>, <string-name name-style="western"><surname>Priyakumar</surname>  <given-names>UD</given-names></string-name></person-group>. <article-title>BiRDS—binding residue detection from protein sequences using deep ResNets</article-title>. <source><italic toggle="yes">J Chem Inf Model</italic></source>  <year>2022</year>;<volume>62</volume>:<fpage>1809</fpage>–<lpage>18</lpage>. <pub-id pub-id-type="doi">10.1021/acs.jcim.1c00972</pub-id><pub-id pub-id-type="pmid">35414182</pub-id>
</mixed-citation></ref><ref id="ref37"><label>37.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author">
<string-name name-style="western">
<surname>Draper</surname>  <given-names>DE</given-names></string-name>, <string-name name-style="western"><surname>Grilley</surname>  <given-names>D</given-names></string-name>, <string-name name-style="western"><surname>Soto</surname>  <given-names>AM</given-names></string-name></person-group>. <article-title>Ions and RNA folding</article-title>. <source><italic toggle="yes">Annu Rev Biophys Biomol Struct</italic></source>  <year>2005</year>;<volume>34</volume>:<fpage>221</fpage>–<lpage>43</lpage>. <pub-id pub-id-type="doi">10.1146/annurev.biophys.34.040204.144511</pub-id><pub-id pub-id-type="pmid">15869389</pub-id>
</mixed-citation></ref><ref id="ref38"><label>38.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author">
<string-name name-style="western">
<surname>Philips</surname>  <given-names>A</given-names></string-name>, <string-name name-style="western"><surname>Milanowska</surname>  <given-names>K</given-names></string-name>, <string-name name-style="western"><surname>Lach</surname>  <given-names>G</given-names></string-name>. <etal>et al.</etal></person-group>  <article-title>MetalionRNA: computational predictor of metal-binding sites in RNA structures</article-title>. <source><italic toggle="yes">Bioinformatics</italic></source>  <year>2012</year>;<volume>28</volume>:<fpage>198</fpage>–<lpage>205</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btr636</pub-id><pub-id pub-id-type="pmid">22110243</pub-id>
<pub-id pub-id-type="pmcid">PMC3259437</pub-id></mixed-citation></ref></ref-list></back></article>