
<!DOCTYPE article
  PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD with MathML3 v1.4 20241031//EN" "JATS-archivearticle1-4-mathml3.dtd">
<article article-type="research-article" xml:lang="en" dtd-version="1.4"><front><journal-meta><journal-id journal-id-type="nlm-ta">Bioinformatics</journal-id><journal-id journal-id-type="iso-abbrev">Bioinformatics</journal-id><journal-id journal-id-type="pmc-domain-id">716</journal-id><journal-id journal-id-type="pmc-domain">bioinfo</journal-id><journal-id journal-id-type="nlm-id">9808944</journal-id><journal-id journal-id-type="publisher-id">bioinformatics</journal-id><journal-title-group><journal-title>Bioinformatics</journal-title></journal-title-group><issn pub-type="ppub">1367-4803</issn><issn pub-type="epub">1367-4811</issn><?publisher_abbrev oup?><publisher><publisher-name>Oxford University Press</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="pmcid">PMC6022650</article-id><article-id pub-id-type="pmcid-ver">PMC6022650.1</article-id><article-id pub-id-type="pmcaid">6022650</article-id><article-id pub-id-type="pmcaiid">6022650</article-id><article-id pub-id-type="pmid">29950017</article-id><article-id pub-id-type="doi">10.1093/bioinformatics/bty263</article-id><article-id pub-id-type="publisher-id">bty263</article-id><article-version article-version-type="pmc-version">1</article-version><article-categories><subj-group subj-group-type="heading"><subject>Ismb 2018–Intelligent Systems for Molecular Biology Proceedings</subject><subj-group subj-group-type="category-toc-heading"><subject>Studies of Phenotypes and Clinical Applications</subject></subj-group></subj-group></article-categories><title-group><article-title>A gene–phenotype relationship extraction pipeline from the biomedical literature using a representation learning approach</article-title></title-group><contrib-group><contrib contrib-type="author"><name name-style="western"><surname>Xing</surname><given-names initials="W">Wenhui</given-names></name><xref ref-type="aff" rid="bty263-aff1">1</xref><xref ref-type="author-notes" rid="bty263-fn1"/></contrib><contrib contrib-type="author"><name name-style="western"><surname>Qi</surname><given-names initials="J">Junsheng</given-names></name><xref ref-type="aff" rid="bty263-aff2">2</xref><xref ref-type="author-notes" rid="bty263-fn1"/></contrib><contrib contrib-type="author"><name name-style="western"><surname>Yuan</surname><given-names initials="X">Xiaohui</given-names></name><xref ref-type="aff" rid="bty263-aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Li</surname><given-names initials="L">Lin</given-names></name><xref ref-type="aff" rid="bty263-aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Zhang</surname><given-names initials="X">Xiaoyu</given-names></name><xref ref-type="aff" rid="bty263-aff3">3</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Fu</surname><given-names initials="Y">Yuhua</given-names></name><xref ref-type="aff" rid="bty263-aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Xiong</surname><given-names initials="S">Shengwu</given-names></name><xref ref-type="aff" rid="bty263-aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Hu</surname><given-names initials="L">Lun</given-names></name><xref ref-type="aff" rid="bty263-aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Peng</surname><given-names initials="J">Jing</given-names></name><xref ref-type="aff" rid="bty263-aff1">1</xref><xref ref-type="corresp" rid="bty263-cor1"/></contrib></contrib-group><aff id="bty263-aff1"><label>1</label>School of Computer Science and Technology, Wuhan University of Technology, Wuhan, China</aff><aff id="bty263-aff2"><label>2</label>Department of Plant Science, College of Biological Science, China Agricultural University, Beijing, China</aff><aff id="bty263-aff3"><label>3</label>Britton Chance Center for Biomedical Photonics, Wuhan National Laboratory for Optoelectronics-Huazhong University of Science and Technology, Wuhan, China</aff><author-notes><corresp id="bty263-cor1">To whom correspondence should be addressed. <email>pengjing@whut.edu.cn</email></corresp><fn id="bty263-fn1"><p>The authors wish it to be known that, in their opinion, Wenhui Xing and Junsheng Qi authors should be regarded as Joint First Authors.</p></fn></author-notes><pub-date pub-type="ppub"><day>01</day><month>7</month><year>2018</year></pub-date><pub-date pub-type="epub" iso-8601-date="2018-06-27"><day>27</day><month>6</month><year>2018</year></pub-date><volume>34</volume><issue>13</issue><issue-id pub-id-type="pmc-issue-id">316001</issue-id><issue-title>ISMB 2018 Proceedings July 6 to July 10, 2018, Chicago, IL, United States</issue-title><fpage>i386</fpage><lpage>i394</lpage><pub-history><event event-type="pmc-release"><date><day>27</day><month>06</month><year>2018</year></date></event><event event-type="pmc-live"><date><day>10</day><month>07</month><year>2018</year></date></event><event event-type="pmc-last-change"><date iso-8601-date="2023-09-26 00:25:09.917"><day>26</day><month>09</month><year>2023</year></date></event></pub-history><permissions><copyright-statement>© The Author(s) 2018. Published by Oxford University Press.</copyright-statement><copyright-year>2018</copyright-year><license xmlns:xlink="http://www.w3.org/1999/xlink" license-type="cc-by-nc" xlink:href="http://creativecommons.org/licenses/by-nc/4.0/"><ali:license_ref xmlns:ali="http://www.niso.org/schemas/ali/1.0/" specific-use="textmining" content-type="ccbynclicense">https://creativecommons.org/licenses/by-nc/4.0/</ali:license_ref><license-p>This is an Open Access article distributed under the terms of the Creative Commons Attribution Non-Commercial License (<ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by-nc/4.0/">http://creativecommons.org/licenses/by-nc/4.0/</ext-link>), which permits non-commercial re-use, distribution, and reproduction in any medium, provided the original work is properly cited. For commercial re-use, please contact journals.permissions@oup.com</license-p></license></permissions><self-uri xmlns:xlink="http://www.w3.org/1999/xlink" content-type="pmc-pdf" xlink:href="bty263.pdf"><?pdf-name bty263.pdf?><?pdf-size 390311?><?pdf-md5 ad9ed3691f64204185ce61e169e6fac9?><?pdf-image-server-status NEVER_LOAD?><?pdf-cloudpmc-urn urn:app:2401/6022650/ad9ed3691f64/bty263.pdf?></self-uri><self-uri xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="bty263.pdf"/><abstract><title>Abstract</title><sec id="s1"><title>Motivation</title><p>The fundamental challenge of modern genetic analysis is to establish gene-phenotype correlations that are often found in the large-scale publications. Because lexical features of gene are relatively regular in text, the main challenge of these relation extraction is phenotype recognition. Due to phenotypic descriptions are often study- or author-specific, few lexicon can be used to effectively identify the entire phenotypic expressions in text, especially for plants.</p></sec><sec id="s2"><title>Results</title><p>We have proposed a pipeline for extracting phenotype, gene and their relations from biomedical literature. Combined with abbreviation revision and sentence template extraction, we improved the unsupervised word-embedding-to-sentence-embedding cascaded approach as representation learning to recognize the various broad phenotypic information in literature. In addition, the dictionary- and rule-based method was applied for gene recognition. Finally, we integrated one of famous information extraction system OLLIE to identify gene-phenotype relations. To demonstrate the applicability of the pipeline, we established two types of comparison experiment using model organism <italic toggle="yes">Arabidopsis thaliana</italic>. In the comparison of state-of-the-art baselines, our approach obtained the best performance (F1-Measure of 66.83%). We also applied the pipeline to 481 full-articles from TAIR gene-phenotype manual relationship dataset to prove the validity. The results showed that our proposed pipeline can cover 70.94% of the original dataset and add 373 new relations to expand it.</p></sec><sec id="s3"><title>Availability and implementation</title><p>The source code is available at <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="http://www.wutbiolab.cn:82/Gene-Phenotype-Relation-Extraction-Pipeline.zip">http://www.wutbiolab.cn: 82/Gene-Phenotype-Relation-Extraction-Pipeline.zip</ext-link>.</p></sec><sec id="s4"><title>Supplementary information</title><p>
<xref ref-type="supplementary-material" rid="sup1">Supplementary data</xref> are available at <italic toggle="yes">Bioinformatics</italic> online.</p></sec></abstract><funding-group><award-group award-type="grant"><funding-source><named-content content-type="funder-name">National Key Research and Development Program</named-content></funding-source><award-id>2016YFD0101900</award-id></award-group><award-group award-type="grant"><funding-source><named-content content-type="funder-name">National Natural Science Foundation of China</named-content><named-content content-type="funder-identifier">10.13039/501100001809</named-content></funding-source><award-id>31701144</award-id></award-group></funding-group><counts><page-count count="9"/></counts><custom-meta-group><custom-meta><meta-name>pmc-status-qastatus</meta-name><meta-value>0</meta-value></custom-meta><custom-meta><meta-name>pmc-status-live</meta-name><meta-value>yes</meta-value></custom-meta><custom-meta><meta-name>pmc-status-embargo</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-status-released</meta-name><meta-value>yes</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-open-access</meta-name><meta-value>yes</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-olf</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-manuscript</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-legally-suppressed</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-has-pdf</meta-name><meta-value>yes</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-has-supplement</meta-name><meta-value>yes</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-pdf-only</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-suppress-copyright</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-is-real-version</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-is-scanned-article</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-preprint</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-in-epmc</meta-name><meta-value>yes</meta-value></custom-meta><custom-meta><meta-name>pmc-license-ref</meta-name><meta-value>CC BY-NC</meta-value></custom-meta></custom-meta-group></article-meta></front><body><sec><title>1 Introduction</title><p>The biomedical literature is vast (<xref rid="bty263-B6" ref-type="bibr">Cohen and Hersh, 2005</xref>), and there is an urgent need to process publications automatically and mine embedded knowledge in the literature to create research hypotheses. Recently, biomedical relationship extraction has gained attention for many downstream text-mining applications, such as event extraction, database creation, knowledge discovery, question answering and decision-making. Natural language processing (NLP) systems have been used for mining special relationships from texts as protein–protein interactions (<xref rid="bty263-B24" ref-type="bibr">Papanikolaou <italic toggle="yes">et al.</italic>, 2015</xref>; <xref rid="bty263-B38" ref-type="bibr">Yang <italic toggle="yes">et al.</italic>, 2011</xref>; <xref rid="bty263-B39" ref-type="bibr">Zhu <italic toggle="yes">et al.</italic>, 2015</xref>), genes and diseases (<xref rid="bty263-B8" ref-type="bibr">Coulet <italic toggle="yes">et al.</italic>, 2010</xref>; <xref rid="bty263-B15" ref-type="bibr">Kim <italic toggle="yes">et al.</italic>, 2017</xref>), drug–drug interactions (<xref rid="bty263-B28" ref-type="bibr">Segura Bedmar <italic toggle="yes">et al.</italic>, 2011</xref>, <xref rid="bty263-B29" ref-type="bibr">2013</xref>), as well as among genes, drugs and mutations (<xref rid="bty263-B3" ref-type="bibr">Cheng <italic toggle="yes">et al.</italic>, 2008</xref>; <xref rid="bty263-B25" ref-type="bibr">Rindflesch <italic toggle="yes">et al.</italic>, 1999</xref>). Such relationship extraction contributes to the development of pharmacogenomics, clinical trial screening and adverse drug reaction identification (<xref rid="bty263-B18" ref-type="bibr">Luo <italic toggle="yes">et al.</italic>, 2017</xref>).</p><p>The central challenge of modern genetic analysis is to establish genotype–phenotype correlations (<xref rid="bty263-B5" ref-type="bibr">Cobb <italic toggle="yes">et al.</italic>, 2013</xref>; <xref rid="bty263-B10" ref-type="bibr">Fu <italic toggle="yes">et al.</italic>, 2014</xref>), which are often found in the biomedical literature, but the volume warrants an automatic and reliable system to extract these information from the text.</p><p>Although relationships have been identified among numerous biological entities, the system for extracting gene–phenotype relationships from the literature is very limited. Regarding species types, the current research focuses more on the relationships between human genes and phenotypes (<xref rid="bty263-B7" ref-type="bibr">Collier <italic toggle="yes">et al.</italic>, 2015</xref>; <xref rid="bty263-B37" ref-type="bibr">Yang <italic toggle="yes">et al.</italic>, 2015</xref>). To our knowledge, there is few such studies for plants. Regarding entity types, research on identifying specific phenotypes such as diseases and gene relationships has received great attention (<xref rid="bty263-B15" ref-type="bibr">Kim <italic toggle="yes">et al.</italic>, 2017</xref>; <xref rid="bty263-B23" ref-type="bibr">Özgür <italic toggle="yes">et al.</italic>, 2008</xref>; <xref rid="bty263-B31" ref-type="bibr">Singhal <italic toggle="yes">et al.</italic>, 2016</xref>). However, text-mining systems that can recognize various phenotype and gene relationships are more difficult and are less robust. The system generally involves annotating raw text with named entities and extracting relationships between these entities. (<xref rid="bty263-B18" ref-type="bibr">Luo <italic toggle="yes">et al.</italic>, 2017</xref>) Named entity recognition (NER) is the foundation of relationship extraction and the effect of entity recognition greatly affects relationship extraction results. (<xref rid="bty263-B4" ref-type="bibr">Chun <italic toggle="yes">et al.</italic>, 2006</xref>) With gene–phenotype relationship extraction, gene and phenotype should be identified. Because lexical features are relatively regular, there are many methods to identify genes in the text. (<xref rid="bty263-B2" ref-type="bibr">Campos <italic toggle="yes">et al.</italic>, 2012</xref>; <xref rid="bty263-B33" ref-type="bibr">Wei <italic toggle="yes">et al.</italic>, 2015</xref>) However, although research on NER has been improved (<xref rid="bty263-B11" ref-type="bibr">Gaizauskas <italic toggle="yes">et al.</italic>, 2003</xref>; <xref rid="bty263-B12" ref-type="bibr">Horn <italic toggle="yes">et al.</italic>, 2004</xref>; <xref rid="bty263-B27" ref-type="bibr">Segura-Bedmar <italic toggle="yes">et al.</italic>, 2008</xref>), phenotype identification is still challenging and this negatively influences relationship extraction.</p><p>First, a phenotype is usually composed of multiple words, such as ‘<italic toggle="yes">calcium sensitivity</italic>’ or ‘<italic toggle="yes">genic male sterility-photoperiod sensitive</italic>’. Thus, name boundaries are complex. Second, phenotypic descriptions are often study- or author-specific due to a lack of standard expressions, complicating this search. For example, in the two sentences ‘…resulting in root growth inhibition, smaller rosettes, and <italic toggle="yes">leaf curling'</italic>. (PMID: 26734017) and ‘…leading to early flowering and <italic toggle="yes">curly leaves phenotypes’</italic>. (PMID: 25693187), the same leaf morphology has two different descriptions, i.e. ‘<italic toggle="yes">leaf curling</italic>’, ‘<italic toggle="yes">curly leaves</italic>’. In addition, while there are specialized lexicons in many areas, no lexicon can be directly used to identify overall phenotypic descriptions in text, especially for plants. For example, the Unified Medical Language System (UMLS) MetaThesaurus (<xref rid="bty263-B13" ref-type="bibr">Humphreys <italic toggle="yes">et al.</italic>, 1998</xref>) is a vocabulary database that includes numerous semantic types, except for <italic toggle="yes">Phenotype</italic> type. In the plant domain, the controlled vocabulary plant trait ontology (PTO) (<ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="http://bioportal.bioontology.org/ontologies/PTO">http://bioportal.bioontology.org/ontologies/PTO</ext-link> ) is too general, so it may not include all species traits. The Arabidopsis Information Resource (TAIR) (<xref rid="bty263-B16" ref-type="bibr">Lamesch <italic toggle="yes">et al.</italic>, 2012</xref>) is curated by manually summarizing published literature so it is limited and difficult to organize for future use. The AraPheno (<xref rid="bty263-B30" ref-type="bibr">Seren <italic toggle="yes">et al.</italic>, 2017</xref>) database is an organization of the Genome-Wide Association Study (GWAS) phenotypic results in only six published studies, so the data are few. These manual curation processes are time-consuming and cannot keep up with rapidly increasing literature.</p><p>Here, we propose a novel gene–phenotype relationship extraction pipeline using model plant <italic toggle="yes">Arabidopsis thaliana</italic>. First we improved the word-embedding-to-sentence-embedding cascaded approach (<xref rid="bty263-B35" ref-type="bibr">Xing <italic toggle="yes">et al.</italic>, 2017</xref>) as representation learning to recognize various broad phenotypic descriptions in large-scale biomolecular literature. Then, genes from the same phenotype-containing sentence were found, using the dictionary-based method. Next, a relationship extraction system Open Language Learning for Information Extraction (OLLIE) was applied to extract gene–phenotype relationships.</p><p>The proposed pipeline improves relationship extractions by identifying more phenotypic descriptions in the text. We identified many types of phenotypic descriptions based on their boundary delimitation: phenotypic phrases and phenotypic long/short sentences. To locate sentences that include the phenotype, we use word embedding to learn distributed representations for words and phrases. Then, we can extract phenotypic phrases missed by ontology, thus extracting more sentences containing phenotypes. Then we cascade the sentence-embedding method for specific phenotype-containing sentences. Due to numerous candidate phenotypic sentences, expert verification is time-consuming. According to the similarity mechanism, we find that sentences with high similarity to the phenotype-containing sentences have similar sentence structures. This prompted us to design a Phenotypic Sentence Template Extraction arChitecture (PSTEC) algorithm that automatically extracts phenotype sentence templates. With these templates, we can extract complex non-phrase forms of long/short phenotypic sentences.</p><p>Ultimately, we evaluated the proposed pipeline from two aspects. (i) We designed three baselines to compare with our proposed relationship extraction pipeline. From the results, we identified more phenotypes (expanding the original ontology almost 3-fold), which significantly improved recall value (improving 24.05% compared to the traditional ontology-based method). Meanwhile, identifying phenotypic descriptions from multiple perspectives also increased the precision of whole recognition. Using the OLLIE system based on machine learning method, we effectively improved F1-Measure compared with traditional relation extraction approach. Thus, our pipeline had a F1-Measure of 66.83%, the greatest of all baselines. (ii) We applied the pipeline to 481 full articles from the TAIR gene–phenotype relationship dataset, and the coverage was 70.94%. Moreover, we added 373 relationships to expand this dataset. Our pipeline automatically identified new relationships with a growing body of literature showing strong scalability. The proposed pipeline is versatile and can be used not only for extraction of relationships in <italic toggle="yes">Arabidopsis</italic> but also for other plant species such as soybean and cotton.</p></sec><sec><title>2 Our gene–phenotype relationship pipeline</title><sec><title>2.1 The overview of our pipeline</title><p>The pipeline starts with scanning abstracts in PubMed using the keyword <italic toggle="yes">‘A.thaliana’</italic> and the Entrez Programming Utilities (E-utilities) web service (<ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/books/NBK25501">https://www.ncbi.nlm.nih.gov/books/NBK25501</ext-link>). We clean irrelevant author information and acquire 63 459 abstracts that mention <italic toggle="yes">A.thaliana</italic>.</p><p>Next, we improve the proposed cascaded representation learning approach (<xref rid="bty263-B35" ref-type="bibr">Xing <italic toggle="yes">et al.</italic>, 2017</xref>) to recognize various broad phenotypes in the literature. Our representation learning approach, combined with the syntactic and semantic analysis of texts, identifies phenotypes in multiple directions from phenotypic phrases to complex short/long phenotypic sentences. Using ontology terms as input, our approach greatly expands the recognition of ontology term synonyms in the literature and establishes a bridge from ontology to literature description, so that study- or author-specific terms can be identified.</p><p>Then we use the results of phenotypic identification to extract gene–phenotype relationships. We use dictionary- and rule-based methods to identify <italic toggle="yes">Arabidopsis</italic> genes in the literature. Then, we combine the workflow of the Open Information Extraction (IE) system with our entity recognition to extract and establish an <italic toggle="yes">Arabidopsis</italic> gene–phenotype binary relationship. The pipeline was implemented and run on a 24 2.4 GHz Xeon core server running on Ubuntu Linux 16.04. <xref ref-type="fig" rid="bty263-F1">Figure 1</xref> shows the overview of the pipeline.
</p><fig id="bty263-F1" orientation="portrait" position="float"><label>Fig. 1.</label><caption><p>The overview of our gene–phenotype relationships extraction pipeline</p></caption><graphic xmlns:xlink="http://www.w3.org/1999/xlink" position="float" orientation="portrait" xlink:href="bty263f1.jpg"><?image-name bty263f1.jpg?><?image-size 64523?><?image-md5 20cc00b5c1adb525c1fccae3e4c88745?><?image-image-server-status LOAD_COMPLETED?><?image-original-height 1057?><?image-original-width 1600?><?image-scaled-height 529?><?image-scaled-width 800?><?image-cloudpmc-urn urn:cdn:blobs/2401/6022650/20cc00b5c1ad/bty263f1.jpg?><?thumb-name bty263f1.gif?><?thumb-size 3577?><?thumb-md5 2f710b9259bd3695c86cf1c451747084?><?thumb-image-server-status NEVER_LOAD?><?thumb-scaled-height 80?><?thumb-scaled-width 121?><?thumb-cloudpmc-urn urn:cdn:blobs/2401/6022650/2f710b9259bd/bty263f1.gif?></graphic></fig></sec><sec><title>2.2 Cascaded approach for phenotype extraction</title><p>Before entity recognition, we used domain-resource ontology to establish the original phenotypic dataset. We extracted phenotypic descriptions from phenotypic phrases and sentences based on different boundaries. We used the parse tree combined with the word embedding method to extract phenotypic phrases, the majority of which were described by noun phrases. Because some synonyms in ontology are not described as phenotype in the text, the previous approach did not consider it leading to some errors. Therefore, we added abbreviation recognition and revision algorithm into the improved cascaded approach.</p><p>Because some special phenotypes are non-phrase forms or long/short sentence descriptions, we used phenotypic sentences from word embedding results as positive samples to cascade the sentence embedding method for finding phenotype sentences. We transformed the unsupervised sentence-embedding model into a weakly supervised model. Due to the lack of training of positive and negative samples, we use the Negative Class Label Enhanced (NCLE) algorithm (<xref rid="bty263-B35" ref-type="bibr">Xing <italic toggle="yes">et al.</italic>, 2017</xref>) to label negative samples and train the sentence-embedding model in combination with the positive samples of the word-embedding results. We analyzed results of sentence embedding, finding that phenotypic sentences gathered by the similarity mechanism had similar structures. However, the previous approach estimated these results through expert verification, which is time-consuming. Therefore, we extracted sentence templates that described the phenotype by improving the algorithm of the statistical combination to expand phenotype recognition.</p><sec><title>2.2.1 Constructing the phenotype dataset</title><p>First, we use two ontologies to create the original phenotype dataset <italic toggle="yes">P</italic>, i.e. PTO and Arabidopsis Hormone Database 2.0 (<ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="http://ahd.cbi.pku.edu.cn/cgi-bin/phenotypeBrowse.pl">http://ahd.cbi.pku.edu.cn/cgi-bin/phenotypeBrowse.pl</ext-link> .) (<xref rid="bty263-B14" ref-type="bibr">Jiang <italic toggle="yes">et al.</italic>, 2011</xref>). PTO is an important controlled vocabulary that describes phenotypic traits in plants. Each trait is a distinguishable, characteristic, quality or phenotypic feature of a developing or mature plant or a plant part. Arabidopsis Hormone Database 2.0 provides a systematic and comprehensive view of genes participating in plant hormonal regulation of the model organism <italic toggle="yes">A. thaliana</italic>. Its phenotypic ontology was developed to describe precisely myriad hormone-regulated morphological processes with standardized vocabularies in <italic toggle="yes">Arabidopsis</italic>.</p><p>When processing PTO, we extract ‘name’ and ‘synonym’ from every term in the ontology. Approximately 84% of these names are associated with synonyms; on average, each name has 1.07 synonyms. For example, the phenotype ‘<italic toggle="yes">alkali soil sensitivity</italic>’ has two synonyms: ‘<italic toggle="yes">AlkS</italic>’ and ‘<italic toggle="yes">alkali sensitivity</italic>’. Not all of terms in these ontologies appear in the literature. We found 805 terms in abstracts after removing duplicate entries. We combined these into a complete phenotype dataset <italic toggle="yes">P</italic>.</p></sec><sec><title>2.2.2 Word embedding</title><p>We followed the word embedding method published in (<xref rid="bty263-B35" ref-type="bibr">Xing <italic toggle="yes">et al.</italic>, 2017</xref>). First, we used the collected PubMed texts to train the word-embedding model, which gave each word or phrase a distributed representation in low and dense dimensional vector space. By finding phrases with high similarity to phenotypic entities in <italic toggle="yes">P</italic>, the original ontology of the phenotype is expanded as <italic toggle="yes">P</italic><sub>update</sub>. Therefore, we can obtain more sentences containing phenotypic information.</p><p>Because some phenotypic synonyms contained in <italic toggle="yes">P</italic> are abbreviated forms, they may not represent as phenotype in the text and are incorrectly identified. For example, the abbreviation ‘<italic toggle="yes">AC</italic>’ in the ontology corresponds to the full name of ‘<italic toggle="yes">leaf sheath auricle color</italic>’. However, in the sentence ‘Many of these proteins have complex domain architectures with AC or GC centers …’ (PMID: 26721677), ‘<italic toggle="yes">AC</italic>’ is not a phenotype. The previous method did not consider abbreviation recognition such as this, so we required post-processing of word-embedding results. After obtaining a high similarity phenotype phrase, we recognized and revised the abbreviation.</p><p>We used (<xref rid="bty263-B36" ref-type="bibr">Xu <italic toggle="yes">et al.</italic>, 2009</xref>) algorithms for identifying abbreviations in the biological literature, matching pairs of all abbreviations and full names in the processed texts. When we used an updated phenotype dataset <italic toggle="yes">P</italic><sub>update</sub> to reidentify the phenotype in the literature, if there was an abbreviated form, it was first matched with a full name. Only the full name of the abbreviation also in <italic toggle="yes">P</italic><sub>update</sub>, remained as a phenotype, otherwise it was deleted. The abbreviation recognition and revision can increase pipeline precision value and identify phenotypes more accurately.</p></sec><sec><title>2.2.3 Sentence embedding</title><p>Using the word-embedding results, we classified and tagged PubMed texts as input for the sentence-embedding (<xref rid="bty263-B17" ref-type="bibr">Le and Mikolov, 2014</xref>) method. The trained model can find sentences containing phenotypic information, acquiring new phenotypic sentences. To improve diversity of phenotype recognition, we transformed the unsupervised sentence-embedding model into a weakly supervised model. We used the results of word-embedding as positive samples, <italic toggle="yes">S</italic><sub>pos</sub>, and combined the NCLE algorithm for negative samples, <italic toggle="yes">S</italic><sub>neg</sub>, for the training of the Sen2Vec model.</p><p>Sentence embedding can aggregate similar phenotypic expressions. We found that large-scale gathered sentences have a similar sentence context structure. For example, the more similar sentences with ‘Solute import across the pollen plasma membrane, which occurs via proteinaceous transporters, <bold>is required to <italic toggle="yes">support pollen development</italic></bold> and also for <bold><italic toggle="yes">subsequent germination and pollen tube growth</italic><italic toggle="yes">’</italic></bold> always have the same structure <italic toggle="yes">‘be required {prep_*} + [phenotype]’</italic>, such as:
<list list-type="bullet"><list-item><p>‘During pollination, constant communication between male pollen and the female stigma <bold>is required for <italic toggle="yes">pollen adhesion, germination, and tube growth</italic></bold>’.</p></list-item><list-item><p>‘Two <italic toggle="yes">A.thaliana</italic> genes, QRT1 and QRT2, <bold>are required for <italic toggle="yes">pollen separation during normal development</italic><italic toggle="yes">’</italic></bold>.</p></list-item></list></p><p>Due to many similar sentences, it is time-consuming to identify all phenotypic sentences and analyze their phenotype with expert evaluation. Therefore, we used sentence structure to automate extraction of complicated long/short phenotypic sentences of non-phrase types. These structures may contain complex phenotypic descriptions, likely with punctuation, prepositions, and conjunctions. We designed an automated algorithm to find frequently occurring sentence templates and with this, we extracted relatively complex descriptions of phenotypic long/short sentences from many sentence-embedding results.</p><p>At present, there are few studies about automatic generation of sentence templates in NLP. We borrowed the idea of modular algorithms from Sentence Pattern Extraction arChitecturte (SPEC) systems in (<xref rid="bty263-B19" ref-type="bibr">Michal <italic toggle="yes">et al.</italic>, 2011</xref>) and proposed our own solution for combinatorial explosion problem.</p><p>With the SPEC algorithm, a ‘sentence template’ is considered as <italic toggle="yes">n</italic>-element ordered combination of sentence elements. It generates all possible combinations of patterns from a sentence and selects the frequency occurrence combination as a sentence pattern. However, we focused on the phenotype-containing structure and created the algorithm Phenotypic Sentence Template Extraction arChitecture (PSTEC) which consists of three components:
<list list-type="order"><list-item><p>Preprocessing</p></list-item><list-item><p>Generation of all ordered combinations from sentence elements</p></list-item><list-item><p>Insertion of a wildcard</p></list-item></list></p><p>
<bold>Preprocessing</bold>: We tokenized all positive sentences <italic toggle="yes">S</italic><sub>pos</sub> of sentence embedding. Because we must extract phenotype-containing sentence structures, we treated phenotypic phrases as a whole and replaced phenotypic descriptions appearing in the sentence with ‘PHE’.</p><p>
<bold>Generation ordered combinations</bold>: In every <italic toggle="yes">n</italic>-element sentence, there is <italic toggle="yes">k</italic>-number of ordered combination groups (1 ≤ <italic toggle="yes">k </italic>≤<italic toggle="yes"> </italic>max). After processing all sentences in corpora, we choose a combination of frequencies greater than a threshold <italic toggle="yes">fre</italic> as a <italic toggle="yes">k</italic>-length template. Because the phenotype-containing template is not too long, so we set max as the length of the element threshold. We set two restrictions to prevent the combination explosions:
<list list-type="bullet"><list-item><p>Combination of the <italic toggle="yes">k</italic>-element must include the specific word ‘PHE’</p></list-item><list-item><p>Any ‘PHE’ contained (<italic toggle="yes">k</italic>-1)-element subset of <italic toggle="yes">k</italic>-element combination must be in the (<italic toggle="yes">k</italic>-1)-element template.</p></list-item></list></p><p>After iteration processing, we obtained all ordered, not duplicated, high frequency combinations for all values of <italic toggle="yes">k</italic> from the range of {1, …, max} as <italic toggle="yes">k</italic>-element sentence templates.</p><p>
<bold>Insertion of a wildcard</bold>: During combination, we combined the original word order. To improve the applicability of templates, we specified whether the elements appeared next to each other or were separated. Therefore, we placed a wildcard between all non-subsequent elements using one heuristic rule. If an absolute difference of word order assigned to the two subsequent elements of a combination &gt;1, we added a wildcard between them. An example of PSTEC algorithm appears in <xref ref-type="fig" rid="bty263-F2">Figure 2</xref>.
</p><fig id="bty263-F2" orientation="portrait" position="float"><label>Fig. 2.</label><caption><p>The procedure for sentence template extraction using high frequency three-element combinations to generate four-element template</p></caption><graphic xmlns:xlink="http://www.w3.org/1999/xlink" position="float" orientation="portrait" xlink:href="bty263f2.jpg"><?image-name bty263f2.jpg?><?image-size 86700?><?image-md5 cff238da6816e251759eb1d0301d8447?><?image-image-server-status LOAD_COMPLETED?><?image-original-height 740?><?image-original-width 950?><?image-scaled-height 493?><?image-scaled-width 633?><?image-cloudpmc-urn urn:cdn:blobs/2401/6022650/cff238da6816/bty263f2.jpg?><?thumb-name bty263f2.gif?><?thumb-size 3668?><?thumb-md5 c4e8c4db4e0320f651b7c4348af55f86?><?thumb-image-server-status NEVER_LOAD?><?thumb-scaled-height 79?><?thumb-scaled-width 102?><?thumb-cloudpmc-urn urn:cdn:blobs/2401/6022650/c4e8c4db4e03/bty263f2.gif?></graphic></fig><p>When we obtained the high-frequency <italic toggle="yes">max</italic>-element sentence templates, we applied these templates to the results of a large number of sentence embeddings. Extracting the description of the more complex phenotypes in sentences that are highly similar to the positive samples improved phenotype recognition.</p></sec></sec><sec><title>2.3 Gene–phenotype relationship extraction</title><p>For gene–phenotype relationship extraction, the gene is required and gene lexical features are relatively regular in texts, gene IDs or gene names may be used to represent them. Therefore, we used a dictionary- and rule-based method to identify genes.</p><p>After entity recognition was complete, our pipeline extracted the relationship with the open information extraction (IE) system. Results of the relationship extraction are expressed as triplets (arg1; r; arg2). The r (relationship phrase) represents arg1 and arg2 entity relationships.</p><sec><title>2.3.1 Gene extraction</title><p>First, we searched all related genes in the UniProt database (<ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="http://www.uniprot.org">http://www.uniprot.org</ext-link> ) using ‘<italic toggle="yes">A.thaliana</italic>’ as a key word and obtained 129 648 records. Each record contained the fields ‘Organism’, ‘Gene locus’, ‘Gene name’. Although we use <italic toggle="yes">Arabidopsis</italic> as a keyword, the results included other species, such as ‘Oryza sativa subsp. japonica (Rice)’. After processing, we obtained 89 287<italic toggle="yes">Arabidopsis</italic> gene ID and gene name pairs and these were used as a dictionary to identify genes.</p><p>Due to the large number of gene names and not a gene locus in the literature, part of the gene name is not in the dictionary. Therefore, we use gene lexical rules and semantic description rules in the text to improve gene recognition. Gene name spelling had some character-level rules as follows:
<list list-type="order"><list-item><p>All capital letters.</p></list-item><list-item><p>A combination of uppercase and lowercase letters.</p></list-item><list-item><p>A combination of numbers, uppercase and lowercase letters.</p></list-item><list-item><p>Those containing hyphens.</p></list-item></list></p><p>Therefore, we used two rule types, mixed character-levels and contextual-levels, to identify the gene. When an input sentence contained these expressions: <italic toggle="yes">Expression of, Accumulation of, Expression levels/patterns of, Targets of, mRNA abundance of, Transcript profiles/levels of</italic>, and the ‘NNP’ (Proper noun, singular) tagged parts in the part-of-speech (POS) tagged sentence complies with our character-level rules, we extracted this special expression as a gene. For example, with the POS tagged sentence: “…HTR4K27Q (‘<italic toggle="yes">NNP</italic>’) overexpression (‘<italic toggle="yes">NN</italic>’) lines (‘<italic toggle="yes">NNS</italic>’) exhibited (‘<italic toggle="yes">VBD</italic>’) deregulated (‘<italic toggle="yes">JJ</italic>’) <bold>expression</bold> (‘<italic toggle="yes">NN</italic>’) <bold>of</bold> (‘<italic toggle="yes">IN</italic>’) <bold><italic toggle="yes">H3K27me3-enriched (‘NNP’)</italic> genes</bold> (‘<italic toggle="yes">NNS</italic>’).” (PMID: 27926813) contains the specific contextual-level description ‘<italic toggle="yes">Expression of</italic> ’, and the ‘NNP’ tagged words satisfy the third and fourth character-level rules. Thus, we can identify gene ‘<bold><italic toggle="yes">H3K27me3-enriched</italic></bold>’.</p><p>Then, we used all sentences that contained the phenotype as input, and the output is two entities that cooccur in sentences. These sentences were used as input to subsequent relationship extraction.</p></sec><sec><title>2.3.2 Relationship extraction</title><p>To the best of our knowledge, there is a limited document annotation corpus of gene–phenotype relationships in <italic toggle="yes">Arabidopsis</italic> species. Currently we are only concerned with gene–phenotype relationships in single sentences. Most relationship recognition systems are not generic and portable so we used the open information extraction (OpenIE) system for this specific relationship identification. OpenIE can extract assertions from massive corpora without a specified vocabulary (<xref rid="bty263-B9" ref-type="bibr">Fader <italic toggle="yes">et al.</italic>, 2011</xref>) from open-domain corpora, such as the Internet and Wikipedia, but in recent years, OpenIE has used biological literature for systematic testing.</p><p>We used an existing OpenIE system, OLLIE (<xref rid="bty263-B26" ref-type="bibr">Schmitz <italic toggle="yes">et al.</italic>, 2012</xref>) as a relationship phrase recognition tool. OLLIE improved several shortcomings of the state-of-the-art system, extracting only relationships mediated by verbs and ignoring context, extracting tuples not asserted as factual. OLLIE is popular for information extraction and used in many fields, such as Question-Answer (<xref rid="bty263-B1" ref-type="bibr">Berant <italic toggle="yes">et al.</italic>, 2013</xref>), knowledge graphs (<xref rid="bty263-B22" ref-type="bibr">Nickel <italic toggle="yes">et al.</italic>, 2016</xref>), and named entities’ network (<xref rid="bty263-B32" ref-type="bibr">Tariq <italic toggle="yes">et al.</italic>, 2017</xref>). OLLIE uses high-precision results of the previous generation OpenIE system i.e. REVERB (<xref rid="bty263-B9" ref-type="bibr">Fader <italic toggle="yes">et al.</italic>, 2011</xref>). With many syntactic analyses of sentences that contain relationships, learning relationship patterns can be extended to find relationships of new input sentences.</p><p>We input the co-occurring sentences into the OLLIE system and extracted relationship sentences and their corresponding relationships. OLLIE automatically gives NP pairs of sentences as arguments in the relationship. However, these NP pairs contain too much noise, and the partially extracted arguments are not genes or phenotypes. Therefore, we limited our screening to eligible relationship groups. For the first (agr1) and third (agr2) parts of one triple, we need map them to the previous phenotype and gene entity list. When one or some genes and phenotypes are in each of the two arguments, we consider the relationship as a gene–phenotype relationship and stored such a relationship.</p></sec></sec></sec><sec><title>3 Results and discussion</title><sec><title>3.1 Phenotype extraction results</title><sec><title>3.1.1 Word-embedding results</title><p>We used Word2Vec (<ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="https://code.google.com/p/word2vec">https://code.google.com/p/word2vec</ext-link> ) to train a skip-gram model with a 4 D size, i.e. 300, 500, 700 and 900. Due to a lack of standards for this topic, we needed expert evaluation and annotation. Therefore, the results of word embedding first were semi-automatically classified and then manually evaluated by one expert and confirmed by another. Ultimately, the word-embedding method can extend original phenotype datasets <italic toggle="yes">P</italic>, increasing 1303 new phenotype data by up to 161.86%. We used the extended dataset <italic toggle="yes">P</italic><sub>update</sub> to match the phenotypic descriptions in the abstracts. Mapping sentences numbered 88 243. After abbreviations were identified and revised, 87 613 sentences containing phenotypes were obtained.</p><p>Some examples of phenotypes recognized by the word-embedding method appear in <xref rid="bty263-T1" ref-type="table">Table 1</xref>. ‘Ontology term’ as the original input, using the similarity mechanism to get ‘Phenotype’ and the corresponding ‘Similarity Score’. ‘Class’ represents the corresponding categories in the PTO 10 basic categories (10 basic categories are: TO: 0000277 biochemical trait; TO: 0000283 biological process trait; TO: 0000183 other miscellaneous trait; TO: 0000357 plant growth and development trait; TO: 0000017 plant morphology trait; TO: 0000597 quality trait; TO: 0000133 stature or vigor trait; TO: 0000392 sterility or fertility trait; TO: 0000164 stress trait; TO: 0000371 yield trait).
<table-wrap id="bty263-T1" orientation="portrait" position="float"><label>Table 1.</label><caption><p>Examples of word-embedding results</p></caption><table frame="hsides" rules="groups"><colgroup span="1"><col valign="top" align="left" span="1"/><col valign="top" align="left" span="1"/><col valign="top" align="char" char="." span="1"/><col valign="top" align="left" span="1"/></colgroup><thead><tr><th align="left" rowspan="1" colspan="1">Ontology term</th><th align="left" rowspan="1" colspan="1">Phenotype</th><th align="left" rowspan="1" colspan="1">Similarity Score</th><th align="left" rowspan="1" colspan="1">Class</th></tr></thead><tbody><tr><td rowspan="4" colspan="1"> Cell elongation</td><td rowspan="1" colspan="1">Cell expansion</td><td rowspan="1" colspan="1">0.671</td><td rowspan="1" colspan="1">TO: 0000357</td></tr><tr><td rowspan="1" colspan="1">Cell enlargement</td><td rowspan="1" colspan="1">0.531</td><td rowspan="1" colspan="1"/></tr><tr><td rowspan="1" colspan="1">Organ expansion</td><td rowspan="1" colspan="1">0.528</td><td rowspan="1" colspan="1"/></tr><tr><td rowspan="1" colspan="1">Cell proliferation</td><td rowspan="1" colspan="1">0.526</td><td rowspan="1" colspan="1"/></tr><tr><td rowspan="4" colspan="1"> Chlorophyll content</td><td rowspan="1" colspan="1">Lower ion leakage</td><td rowspan="1" colspan="1">0.625</td><td rowspan="1" colspan="1">TO: 0000277</td></tr><tr><td rowspan="1" colspan="1">Photosystem II activity</td><td rowspan="1" colspan="1">0.557</td><td rowspan="1" colspan="1"/></tr><tr><td rowspan="1" colspan="1">Photosynthetic quantum yield</td><td rowspan="1" colspan="1">0.550</td><td rowspan="1" colspan="1"/></tr><tr><td rowspan="1" colspan="1">Higher relative water content</td><td rowspan="1" colspan="1">0.531</td><td rowspan="1" colspan="1"/></tr><tr><td rowspan="4" colspan="1"> Chloroplast structure</td><td rowspan="1" colspan="1">Photosynthetic phenotype</td><td rowspan="1" colspan="1">0.498</td><td rowspan="1" colspan="1">TO: 0000017</td></tr><tr><td rowspan="1" colspan="1">Thylakoid structure</td><td rowspan="1" colspan="1">0.495</td><td rowspan="1" colspan="1"/></tr><tr><td rowspan="1" colspan="1">Leaf chloroplast ultrastructure</td><td rowspan="1" colspan="1">0.484</td><td rowspan="1" colspan="1"/></tr><tr><td rowspan="1" colspan="1">Pale green leaves</td><td rowspan="1" colspan="1">0.479</td><td rowspan="1" colspan="1"/></tr><tr><td rowspan="4" colspan="1"> Leaf curling</td><td rowspan="1" colspan="1">Dark green leaves</td><td rowspan="1" colspan="1">0.613</td><td rowspan="1" colspan="1">TO: 0000357</td></tr><tr><td rowspan="1" colspan="1">Altered leaf shape</td><td rowspan="1" colspan="1">0.581</td><td rowspan="1" colspan="1"/></tr><tr><td rowspan="1" colspan="1">Curly leaves</td><td rowspan="1" colspan="1">0.576</td><td rowspan="1" colspan="1"/></tr><tr><td rowspan="1" colspan="1">Serrated leaves</td><td rowspan="1" colspan="1">0.558</td><td rowspan="1" colspan="1"/></tr><tr><td rowspan="4" colspan="1"> Drought sensitivity</td><td rowspan="1" colspan="1">Reduced water loss</td><td rowspan="1" colspan="1">0.550</td><td rowspan="1" colspan="1">TO: 0000164</td></tr><tr><td rowspan="1" colspan="1">Enhanced drought resistance</td><td rowspan="1" colspan="1">0.544</td><td rowspan="1" colspan="1"/></tr><tr><td rowspan="1" colspan="1">Drought stress tolerance</td><td rowspan="1" colspan="1">0.539</td><td rowspan="1" colspan="1"/></tr><tr><td rowspan="1" colspan="1">Reduced drought tolerance</td><td rowspan="1" colspan="1">0.535</td><td rowspan="1" colspan="1"/></tr></tbody></table><table-wrap-foot><fn id="tblfn1"><p><italic toggle="yes">Note:</italic> According to the original ‘Ontology term’, we use similarity mechanisms to extract ‘Phenotype’ and its corresponding ‘Similarity Score’. ‘Class’ represents the corresponding categories in the PTO 10 basic categories.</p></fn></table-wrap-foot></table-wrap></p><p>As shown in <xref rid="bty263-T1" ref-type="table">Table 1</xref>, the word-embedding method can find a phenotypic description according to the syntax and context of the text. For example, for the same ontology term ‘<italic toggle="yes">leaf curling</italic>’ (TO: 0002681), the method can extract similar words by considering syntax (‘<italic toggle="yes">leaf curling</italic>’—‘<italic toggle="yes">curly leaves</italic>’) and context semantics (‘<italic toggle="yes">leaf curling</italic>’—‘<italic toggle="yes">altered leaf shape</italic>’). Some new phenotypes are not synonyms of their corresponding original ontology terms. For example, the new phenotype ‘<italic toggle="yes">serrated leaves</italic>’ is not synonymous with ‘<italic toggle="yes">leaf curling</italic>’. This may because the contextual environment that describes the new phenotype and the original term is similar, but the semantics of expression are not the same.</p></sec><sec><title>3.1.2 Sentence-embedding results</title><p>We used Doc2Vec (<ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="http://radimrehurek.com/gensim/models/doc2vec.html">http://radimrehurek.com/gensim/models/doc2vec.html</ext-link> ) to train the PV-DBOW model, and the trained corpora are positive/negative labeled abstracts. Then, we used the results of word embedding <italic toggle="yes">S</italic><sub>pos</sub> as inputs and acquired candidate sentences with similarities greater than <italic toggle="yes">Sim</italic> after calculating for cosine distance with <italic toggle="yes">S</italic><sub>pub</sub>. A reasonable <italic toggle="yes">Sim</italic> value greatly influenced the results. After testing, if <italic toggle="yes">Sim</italic> was too high (&gt;0.4), high similarity sentences were too few and an average of 1.2 high-similarity sentences was obtained for each original sentence. If <italic toggle="yes">Sim</italic> was too low (¡0.2), we get a lot of dissimilar sentences. Therefore, we set <italic toggle="yes">Sim</italic> as 0.3, and an average of 4.5 high-similarity sentences was obtained for each original sentence.</p><p>The sentence-embedding method can find many candidate phenotypic sentences, which contain many non-phrase, complex long/short phenotypic sentences. For example, the phenotypic structure ‘<italic toggle="yes">response to …stress</italic>’ in the sentence ‘GmaPHO1 genes had altered expression in <bold><italic toggle="yes">response to salt, osmotic, and inorganic phosphate stresses</italic><italic toggle="yes">’</italic></bold>. Such phenotypic descriptions are special and numerous and can improve relationship identification. Therefore, we designed a PSTEC algorithm to automatically generate phenotypic sentence templates for extracting them.</p><p>We tested and selected template length <italic toggle="yes">max</italic> and template frequency <italic toggle="yes">fre</italic> of the PSTEC algorithm. When <italic toggle="yes">max</italic> is too long (&gt;6), the template will contain a lot of noise, such as too many prepositions and stop words. When <italic toggle="yes">max</italic> is too short (&lt;4), the template cannot contain complete template structure information. Thus, we set <italic toggle="yes">max</italic> as 5. The size of <italic toggle="yes">fre</italic> directly affects the efficiency and uptime of the algorithm. After testing, we set <italic toggle="yes">fre</italic> as 100 and only kept templates that appeared more often than 100 in the corpus. Ultimately, we obtained 250 sentence templates. There are many types of duplicate templates and high frequency but not intention-containing templates, such as <italic toggle="yes">‘Show/Suggest + prep_*’</italic>. Therefore, we merged and selected these results. <xref rid="bty263-T2" ref-type="table">Table 2</xref> shows 5 high frequency sentence templates that can recognize combination type phenotypes (‘<italic toggle="yes">Tolerance to salt/drought/methyl viologen stress in Arabidopsis’</italic>), with environmental or time factors (‘<italic toggle="yes">Hypocotyl growth in response to unilateral blue-light illumination</italic>’) and are rich in diversity of phenotypes. Meanwhile, we noticed that phenotypes recognized by different templates may differ. For example, the ‘Respond’ template can identify more ‘stress trait’ types.
<table-wrap id="bty263-T2" orientation="portrait" position="float"><label>Table 2.</label><caption><p>Examples of sentence templates</p></caption><table frame="hsides" rules="groups"><colgroup span="1"><col valign="top" align="left" span="1"/><col valign="top" align="left" span="1"/><col valign="top" align="char" char="." span="1"/></colgroup><thead><tr><th align="left" rowspan="1" colspan="1">Sentence template</th><th align="left" rowspan="1" colspan="1">Example of phenotype</th><th align="left" rowspan="1" colspan="1">Number of phenotype</th></tr></thead><tbody><tr><td rowspan="1" colspan="1">Inhibition of + (PHE)</td><td rowspan="1" colspan="1">Root growth the root-swelling phenotype; Germination and elongation of <italic toggle="yes">Arabidopsis</italic> seedling</td><td rowspan="1" colspan="1">127</td></tr><tr><td rowspan="1" colspan="1">Involve(d) in + (PHE)</td><td rowspan="1" colspan="1">Host cell death in the hypersensitive disease-resistance response; <italic toggle="yes">A. thaliana</italic> seedling root to a rapid change in salinity</td><td rowspan="1" colspan="1">532</td></tr><tr><td rowspan="1" colspan="1">(Play a/an adj./n.) Role in + (PHE)</td><td rowspan="1" colspan="1">Coordinate the directional growth of plant tissue; Tolerance to salt/drought/methyl viologen stress in <italic toggle="yes">Arabidopsis</italic></td><td rowspan="1" colspan="1">243</td></tr><tr><td rowspan="1" colspan="1">Regulator/regulation of + (PHE)</td><td rowspan="1" colspan="1">Secondary wall synthesis in fiber of <italic toggle="yes">A.thaliana</italic> stem; Stomatal clustering and density early in <italic toggle="yes">Arabidopsis</italic> leaf development</td><td rowspan="1" colspan="1">197</td></tr><tr><td rowspan="1" colspan="1">(In) Response to + (PHE)</td><td rowspan="1" colspan="1">Both high- and low-temperature stress; Signal emanate from cell undergo pathogen-induced hypersensitive cell death</td><td rowspan="1" colspan="1">215</td></tr></tbody></table><table-wrap-foot><fn id="tblfn2"><p><italic toggle="yes">Note</italic>: PHE represents phenotype, parentheses indicate optional parts.</p></fn></table-wrap-foot></table-wrap></p><p>We can extend 1314 phenotypic descriptions using the sentence template. After merged results of word-embedding, we expanded 2409 phenotypic expressions and increased them 2.99-fold compared to the original phenotype dataset <italic toggle="yes">P</italic>.</p></sec></sec><sec><title>3.2 Gene–phenotype relationship results</title><p>We evaluated results of gene–phenotype extraction from two perspectives.
<list list-type="order"><list-item><p>According to different phenotype recognition and relation extraction methods, we compared with baselines.</p></list-item><list-item><p>We used the entire pipeline in the TAIR database, which manually extracted gene–phenotype relationships from 555 full papers.</p></list-item></list></p><sec><title>3.2.1 Performance comparison with baselines</title><p>Using the phenotype recognition cascaded approach can improve the identification of phenotypes in the literature and improve relationship identification. To illustrate the importance of phenotypic recognition in relationship extraction and to verify the accuracy of our approach, we establish two baselines for performance comparison.
<list list-type="bullet"><list-item><p>B1: Using the traditional ontology-based method (<xref rid="bty263-B21" ref-type="bibr">Müller <italic toggle="yes">et al.</italic>, 2004</xref>) to recognize phenotype and extracting the gene and relationship using method described in this article.</p></list-item><list-item><p>B2: Using the ontology-based with word embedding method (<xref rid="bty263-B20" ref-type="bibr">Mikolov <italic toggle="yes">et al.</italic>, 2013</xref>) to recognize phenotype and extracting the gene and relationship using method described in this article. we also compare with another baseline that use traditional relation extraction methods.</p></list-item><list-item><p>B3: Using method described in this article to extract phenotype and gene, the relation extraction method is based on syntatic rules (<xref rid="bty263-B8" ref-type="bibr">Coulet <italic toggle="yes">et al.</italic>, 2010</xref>) which uses the collapsed dependencies graph representation.</p></list-item></list></p><p>We randomly selected 100 abstracts to identify the relationships by expert verification and to calculate Precision, Recall, F1-Measure. Results are shown in <xref rid="bty263-T3" ref-type="table">Table 3</xref>.
<table-wrap id="bty263-T3" orientation="portrait" position="float"><label>Table 3.</label><caption><p>Performance of baselines compared with our pipeline</p></caption><table frame="hsides" rules="groups"><colgroup span="1"><col valign="top" align="left" span="1"/><col valign="top" align="left" span="1"/><col valign="top" align="left" span="1"/><col valign="top" align="char" char="." span="1"/><col valign="top" align="char" char="." span="1"/><col valign="top" align="char" char="." span="1"/></colgroup><thead><tr><th align="left" rowspan="1" colspan="1">Type</th><th align="left" rowspan="1" colspan="1">Phenotype extraction</th><th align="left" rowspan="1" colspan="1">Relation extraction</th><th align="left" rowspan="1" colspan="1">Precision (%)</th><th align="left" rowspan="1" colspan="1">Recall (%)</th><th align="left" rowspan="1" colspan="1">F1-Measure (%)</th></tr></thead><tbody><tr><td rowspan="1" colspan="1">B1</td><td rowspan="1" colspan="1">Ontology-based (<xref rid="bty263-B21" ref-type="bibr">Müller <italic toggle="yes">et al.</italic>, 2004</xref>)</td><td rowspan="1" colspan="1">OLLIE</td><td rowspan="1" colspan="1">52.98</td><td rowspan="1" colspan="1">33.76</td><td rowspan="1" colspan="1">41.24</td></tr><tr><td rowspan="1" colspan="1">B2</td><td rowspan="1" colspan="1">Ontology-based + word embedding (<xref rid="bty263-B20" ref-type="bibr">Mikolov <italic toggle="yes">et al.</italic>, 2013</xref>)</td><td rowspan="1" colspan="1">OLLIE</td><td rowspan="1" colspan="1">73.91</td><td rowspan="1" colspan="1">50.21</td><td rowspan="1" colspan="1">59.80</td></tr><tr><td rowspan="1" colspan="1">B3</td><td rowspan="1" colspan="1">Representation learning approach</td><td rowspan="1" colspan="1">Syntatic rules  (<xref rid="bty263-B8" ref-type="bibr">Coulet <italic toggle="yes">et al.</italic>, 2010</xref>)</td><td rowspan="1" colspan="1">55.75</td><td rowspan="1" colspan="1">26.58</td><td rowspan="1" colspan="1">36.00</td></tr><tr><td rowspan="1" colspan="1">Our pipeline</td><td rowspan="1" colspan="1">Representation learning approach</td><td rowspan="1" colspan="1">OLLIE</td><td rowspan="1" colspan="1"><bold>7</bold>9.19</td><td rowspan="1" colspan="1"><bold>5</bold>7.81</td><td rowspan="1" colspan="1"><bold>6</bold>6.83</td></tr></tbody></table></table-wrap></p><p>Among the different methods on phenotype recognition, the effect of recognizing the gene–phenotype relationship using only ontology-based efforts is the poorest. Because of loss of many phenotypes, recall value in relationship recognition is low. For example, the phenotype ‘<italic toggle="yes">NaCl stress-sensitive phenotype</italic>’ is not in ontology, so the relationship (MCK1; complemented; <italic toggle="yes">NaCl stress-sensitive phenotype</italic>) cannot be found. However, we can identify this phenotype using the proposed approach and obtain relationships with the best recall. This is because we recognized the phenotypic phrase and the more complex phenotypic long/short sentences based on the sentence template. As the integrity of the phenotype increased, the precision is improved. For this sentence, ‘…a structurally related Arabidopsis MADS-box gene involved in the <italic toggle="yes">negative control of Arabidopsis flowering time</italic>, …’ (PMID: 15539492), due to the template: ‘<italic toggle="yes">(gene) involve + {prep.} + PHE</italic>’, we can identify the whole phenotypic description ‘<italic toggle="yes">negative control of Arabidopsis flowering time</italic>’, and get the relationship (MADS-box gene; involved in; <italic toggle="yes">negative control of Arabidopsis flowering time</italic>). However, the first two baselines only extracted part of the whole expression ‘<italic toggle="yes">flowering time</italic>’ and missed the complete relationships. Thus, our approach can extend relationship extraction by improving phenotype recognition.</p><p>Compared with the B3 baselines, which only change the relation extraction method, our pipleline also has the best performance. Because the syntactic rule method misses many results and only getting 26.58% recall value, its F1-Measure is about 36.00%.</p><p>We have considered to use generic tools such as GNormPlus (<xref rid="bty263-B33" ref-type="bibr">Wei <italic toggle="yes">et al.</italic>, 2015</xref>), GenNorm (<xref rid="bty263-B34" ref-type="bibr">Wei and Kao, 2011</xref>) and so on for gene identification but found that these tools identify the gene of all species that appear in the text. Therefore, noise information is mixed in the targeted identification of <italic toggle="yes">Arabidopsis</italic> gene information, which requires expert screening. So, we finally chose a more targeted rule- and dictionary-based approach and obtained 88.76% precision value in the above test dataset. This is slightly higher than the results given in the article (<xref rid="bty263-B33" ref-type="bibr">Wei <italic toggle="yes">et al.</italic>, 2015</xref>) by GNormPlus (precision 87.1%) and GenNorm (precision 78.9%).</p><p>Although the proposed pipeline can improve the effectiveness of final relationship identification compared with baselines, there are misidentifications and omissions due to the following reasons:
<list list-type="order"><list-item><p>Error of relationship recognition. The OLLIE system is limited as it can only identify the relationship in a single sentence, and the length of the sentence cannot be too long. Sentences &gt;20 words have increased errors for relationship analysis (<xref rid="bty263-B26" ref-type="bibr">Schmitz <italic toggle="yes">et al.</italic>, 2012</xref>). For example, the sentence ‘Hence, the narrow organ shape, reduced plant height, and reduced whorl 4 organ primordia are consistent with a general reduction of cell number, and, perhaps, reflect a role of SEU in promoting cell proliferation’ can be assessed by OLLIE to get this relationship (whorl 4 organ primordia; perhaps reflect; a role SEU in promoting cell proliferation). The wrong relationship association results in inaccurate identification of it.</p></list-item><list-item><p>Inaccurate phenotypic boundary. Although we can identify phenotype from phrases and long/short sentences, more complex phenotypes cause errors or incomplete identification. For example ‘The AGAMOUS gene of Arabidopsis is necessary for the <bold><italic toggle="yes">proper development of stamens and carpels and the prevention of indeterminate growth of the floral meristem</italic><italic toggle="yes">’</italic></bold>. We did not recognize this sentence structure, resulting in incomplete recognition of relationships.</p></list-item><list-item><p>Problem of gene recognition. Although we use a relatively complete <italic toggle="yes">Arabidopsis</italic> gene database as a dictionary for gene ID and gene name identification, and get high precition value of 88.76%, the database may still missing some gene name as well as the corresponding relationship for it.</p></list-item></list></p><p>These errors reduce precision and recall because each case results in an incorrect or incomplete relationship extraction.</p></sec><sec><title>3.2.2 Comparison with TAIR</title><p>The TAIR database (<xref rid="bty263-B16" ref-type="bibr">Lamesch <italic toggle="yes">et al.</italic>, 2012</xref>) is one of the most informative databases for storing <italic toggle="yes">Arabidopsis</italic> information, which contains a gene–phenotype relationships dataset. This information was manually extracted from 555 full texts. To verify pipeline effectiveness, we calculated coverage of relationship for these papers. Because some documents cannot be downloaded, we retrieved only 481 full papers. Preprocessing the TAIR dataset by deleting irrelevant fields, i.e. ‘Phenotype not described’ and ‘No visible phenotype’ was done and we retrieved 1397 sets of gene–phenotype relationships. We noticed that there are duplicate types of relationships in the dataset. For example, the gene ‘MSSP1’ is related with:
<list list-type="bullet"><list-item><p>Under normal growth temperature conditions, the double mutant leaves’ content in glucose and fructose is slightly reduced (30%) in a similar fashion to that observed with the tmt1 single mutants.</p></list-item><list-item><p>Under normal temperature conditions, a substantial reduction in glucose and fructose contents in leaves is observed compared to wild type, and even the single tmt1 and double tmt1/tmt2 mutants.</p></list-item></list></p><p>As they are the same type, we treat them as the identical relationships. We applied our pipeline to this dataset, extracted data were compared with processed TAIR datasets by four experts and offered coverage of 70.94%. Moreover, our pipeline can identify 373 new relationships, which the TAIR dataset does not include. The results are shown in <xref ref-type="supplementary-material" rid="sup1">Supplementary Material</xref>. We had limited coverage for a few reasons:
<list list-type="order"><list-item><p>Many relationships in TAIR come from cross-sentence or even cross-paragraph relationships. Such relationships are unrecognizable to our pipeline that only extracts from a single sentence, so there is the main reason of limited coverage. However, due to redundancy of much information, our pipeline use repetitive relationships extracted from many studies to compensate extraction of the relationship representations in small samples. Such work cannot be done manually.</p></list-item><list-item><p>Many phenotypes in TAIR have not been described in the original literature after subsequent manual processing and summary and this will influence coverage.</p></list-item><list-item><p>There is only a gene locus name in the TAIR dataset, but most documents only describe the gene name. Some gene loci in the gene database do not have corresponding names. Thus, our pipeline cannot recognize these genes or any corresponding relationships.</p></list-item></list></p><p>After analysis, we found that articles in the TAIR dataset are relatively old (most prior to 2000). Due to limitations to manual reading, this dataset failed to update gene–phenotype relationships as the literature grew, so scalability was poor. However, with our pipeline we can quickly find relationships for updated literature, greatly improving efficiency for summarizing data.</p></sec></sec></sec><sec><title>4 Conclusion and future works</title><p>Much plant gene–phenotypic information exists in the biomedical literature, and it continues to grow. Thus, we propose a pipeline to extract relationships between genes and phenotypes using <italic toggle="yes">A.thaliana</italic> as an experimental object. Our pipeline can expand the expression of original phenotype ontology terms in the literature using an improved cascaded representation learning approach of phenotype recognition. This can enhance relationship extraction. Our pipeline obtained an F1-score (66.83%) that outperformed other baselines. Applying the pipeline to the TAIR dataset, we can complement 373 new relationships.</p><p>Future studies may include considering environmental influences and phenotypic conditions for constructing gene–phenotype event extraction instead of binary relationships. If the division of phenotype and relationship boundaries is more detailed, performance will be improved.</p></sec><sec><title>Funding</title><p>This work was supported by National Key Research and Development Program [grant no. 2016YFD0101900], National Natural Science Foundation of China [grant no. 31701144].</p><p>
<italic toggle="yes">Conflict of Interest</italic>: none declared.</p></sec><sec sec-type="supplementary-material"><title>Supplementary Material</title><supplementary-material content-type="local-data" id="sup1" position="float" orientation="portrait"><label>Supplementary Data</label><media xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="bty263_peng.23.sup.1.pdf" position="float" orientation="portrait"><?suppdata-name bty263_peng.23.sup.1.pdf?><?suppdata-size 1155454?><?suppdata-md5 695e4d55856b4960b3f09945a6a27b90?><?suppdata-image-server-status NEVER_LOAD?><?suppdata-mime-type application?><?suppdata-mime-sub-type pdf?><?suppdata-cloudpmc-urn urn:app:2401/6022650/695e4d55856b/bty263_peng.23.sup.1.pdf?><caption><p>Click here for additional data file.</p></caption></media></supplementary-material></sec></body><back><ref-list><title>References</title><ref id="bty263-B1"><mixed-citation publication-type="other">
<person-group person-group-type="author"><name name-style="western"><surname>Berant</surname><given-names>J.</given-names></name></person-group>
<etal>et al</etal> (<year>2013</year>) Semantic parsing on freebase from question-answer pairs. In: <italic toggle="yes">EMNLP</italic>, Vol. 2, p. 6.</mixed-citation></ref><ref id="bty263-B2"><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Campos</surname><given-names>D.</given-names></name></person-group>
<etal>et al</etal> (<year>2012</year>) 
<article-title>Harmonization of gene/protein annotations: towards a gold standard medline</article-title>. <source>Bioinformatics</source>, <volume>28</volume>, <fpage>1253</fpage>–<lpage>1261</lpage>.<pub-id pub-id-type="pmid">22419783</pub-id><pub-id pub-id-type="doi" assigning-authority="pmc">10.1093/bioinformatics/bts125</pub-id></mixed-citation></ref><ref id="bty263-B3"><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Cheng</surname><given-names>D.</given-names></name></person-group>
<etal>et al</etal> (<year>2008</year>) 
<article-title>Polysearch: a web-based text mining system for extracting relationships between human diseases, genes, mutations, drugs and metabolites</article-title>. <source>Nucleic Acids Res</source>., <volume>36</volume>, <fpage>W399</fpage>–<lpage>W405</lpage>.<pub-id pub-id-type="pmid">18487273</pub-id><pub-id pub-id-type="doi" assigning-authority="pmc">10.1093/nar/gkn296</pub-id><pub-id pub-id-type="pmcid">PMC2447794</pub-id></mixed-citation></ref><ref id="bty263-B4"><mixed-citation publication-type="other">
<person-group person-group-type="author"><name name-style="western"><surname>Chun</surname><given-names>H.-W.</given-names></name></person-group>
<etal>et al</etal> (<year>2006</year>) Extraction of gene-disease relations from medline using domain dictionaries and machine learning. In: <italic toggle="yes">Pacific Symposium on Biocomputing</italic>, Vol. 11, Big Island, Hawaii, pp. <fpage>4</fpage>–<lpage>15</lpage>.<pub-id pub-id-type="pmid">17094223</pub-id></mixed-citation></ref><ref id="bty263-B5"><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Cobb</surname><given-names>J.N.</given-names></name></person-group>
<etal>et al</etal> (<year>2013</year>) 
<article-title>Next-generation phenotyping: requirements and strategies for enhancing our understanding of genotype–phenotype relationships and its relevance to crop improvement</article-title>. <source>Theor. Appl. Genet</source>., <volume>126</volume>, <fpage>867</fpage>–<lpage>887</lpage>.<pub-id pub-id-type="pmid">23471459</pub-id><pub-id pub-id-type="doi" assigning-authority="pmc">10.1007/s00122-013-2066-0</pub-id><pub-id pub-id-type="pmcid">PMC3607725</pub-id></mixed-citation></ref><ref id="bty263-B6"><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Cohen</surname><given-names>A.M.</given-names></name>, <name name-style="western"><surname>Hersh</surname><given-names>W.R.</given-names></name></person-group> (<year>2005</year>) 
<article-title>A survey of current work in biomedical text mining</article-title>. <source>Brief. Bioinformatics</source>, <volume>6</volume>, <fpage>57</fpage>–<lpage>71</lpage>.<pub-id pub-id-type="pmid">15826357</pub-id><pub-id pub-id-type="doi" assigning-authority="pmc">10.1093/bib/6.1.57</pub-id></mixed-citation></ref><ref id="bty263-B7"><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Collier</surname><given-names>N.</given-names></name></person-group>
<etal>et al</etal> (<year>2015</year>) 
<article-title>Phenominer: from text to a database of phenotypes associated with OMIM diseases</article-title>. <source>Database</source>, <volume>2015</volume>, bav104. <pub-id pub-id-type="doi" assigning-authority="pmc">10.1093/database/bav104</pub-id><pub-id pub-id-type="pmcid">PMC4622021</pub-id><pub-id pub-id-type="pmid">26507285</pub-id></mixed-citation></ref><ref id="bty263-B8"><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Coulet</surname><given-names>A.</given-names></name></person-group>
<etal>et al</etal> (<year>2010</year>) 
<article-title>Using text to build semantic networks for pharmacogenomics</article-title>. <source>J. Biomed. Informatics</source>, <volume>43</volume>, <fpage>1009</fpage>–<lpage>1019</lpage>.<pub-id pub-id-type="doi" assigning-authority="pmc">10.1016/j.jbi.2010.08.005</pub-id><pub-id pub-id-type="pmcid">PMC2991587</pub-id><pub-id pub-id-type="pmid">20723615</pub-id></mixed-citation></ref><ref id="bty263-B9"><mixed-citation publication-type="other">
<person-group person-group-type="author"><name name-style="western"><surname>Fader</surname><given-names>A.</given-names></name></person-group>
<etal>et al</etal> (<year>2011</year>) Identifying relations for open information extraction. In <italic toggle="yes">Proceedings of the Conference on Empirical Methods in Natural Language Processing</italic>, pp. <fpage>1535</fpage>–<lpage>1545</lpage>. Association for Computational Linguistics.</mixed-citation></ref><ref id="bty263-B10"><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Fu</surname><given-names>R.</given-names></name></person-group>
<etal>et al</etal> (<year>2014</year>) 
<article-title>Genotype–phenotype correlations in neurogenetics: lesch-nyhan disease as a model disorder</article-title>. <source>Brain</source>, <volume>137</volume>, <fpage>1282</fpage>–<lpage>1303</lpage>.<pub-id pub-id-type="pmid">23975452</pub-id><pub-id pub-id-type="doi" assigning-authority="pmc">10.1093/brain/awt202</pub-id><pub-id pub-id-type="pmcid">PMC3999711</pub-id></mixed-citation></ref><ref id="bty263-B11"><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Gaizauskas</surname><given-names>R.</given-names></name></person-group>
<etal>et al</etal> (<year>2003</year>) 
<article-title>Protein structures and information extraction from biological texts: the pasta system</article-title>. <source>Bioinformatics</source>, <volume>19</volume>, <fpage>135</fpage>–<lpage>143</lpage>.<pub-id pub-id-type="pmid">12499303</pub-id><pub-id pub-id-type="doi" assigning-authority="pmc">10.1093/bioinformatics/19.1.135</pub-id></mixed-citation></ref><ref id="bty263-B12"><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Horn</surname><given-names>F.</given-names></name></person-group>
<etal>et al</etal> (<year>2004</year>) 
<article-title>Automated extraction of mutation data from the literature: application of mutext to g protein-coupled receptors and nuclear hormone receptors</article-title>. <source>Bioinformatics</source>, <volume>20</volume>, <fpage>557</fpage>–<lpage>568</lpage>.<pub-id pub-id-type="pmid">14990452</pub-id><pub-id pub-id-type="doi" assigning-authority="pmc">10.1093/bioinformatics/btg449</pub-id></mixed-citation></ref><ref id="bty263-B13"><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Humphreys</surname><given-names>B.L.</given-names></name></person-group>
<etal>et al</etal> (<year>1998</year>) 
<article-title>The unified medical language system: an informatics research collaboration</article-title>. <source>J. Am. Med. Informatics Assoc</source>., <volume>5</volume>, <fpage>1</fpage>–<lpage>11</lpage>.<pub-id pub-id-type="doi" assigning-authority="pmc">10.1136/jamia.1998.0050001</pub-id><pub-id pub-id-type="pmcid">PMC61271</pub-id><pub-id pub-id-type="pmid">9452981</pub-id></mixed-citation></ref><ref id="bty263-B14"><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Jiang</surname><given-names>Z.</given-names></name></person-group>
<etal>et al</etal> (<year>2011</year>) 
<article-title>Ahd2. 0: an update version of arabidopsis hormone database for plant systematic studies</article-title>. <source>Nucleic Acids Res</source>., <volume>39</volume>, <fpage>D1123</fpage>–<lpage>D1129</lpage>.<pub-id pub-id-type="pmid">21045062</pub-id><pub-id pub-id-type="doi" assigning-authority="pmc">10.1093/nar/gkq1066</pub-id><pub-id pub-id-type="pmcid">PMC3013673</pub-id></mixed-citation></ref><ref id="bty263-B15"><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Kim</surname><given-names>J.</given-names></name></person-group>
<etal>et al</etal> (<year>2017</year>) 
<article-title>An analysis of disease-gene relationship from medline abstracts by digsee</article-title>. <source>Sci. Rep</source>., <volume>7</volume>, <fpage>40154.</fpage><pub-id pub-id-type="pmid">28054646</pub-id><pub-id pub-id-type="doi" assigning-authority="pmc">10.1038/srep40154</pub-id><pub-id pub-id-type="pmcid">PMC5215527</pub-id></mixed-citation></ref><ref id="bty263-B16"><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Lamesch</surname><given-names>P.</given-names></name></person-group>
<etal>et al</etal> (<year>2012</year>) 
<article-title>The arabidopsis information resource (tair): improved gene annotation and new tools</article-title>. <source>Nucleic Acids Res</source>., <volume>40</volume>, <fpage>D1202</fpage>–<lpage>D1210</lpage>.<pub-id pub-id-type="pmid">22140109</pub-id><pub-id pub-id-type="doi" assigning-authority="pmc">10.1093/nar/gkr1090</pub-id><pub-id pub-id-type="pmcid">PMC3245047</pub-id></mixed-citation></ref><ref id="bty263-B17"><mixed-citation publication-type="other">
<person-group person-group-type="author"><name name-style="western"><surname>Le</surname><given-names>Q.</given-names></name>, <name name-style="western"><surname>Mikolov</surname><given-names>T.</given-names></name></person-group> (<year>2014</year>) Distributed representations of sentences and documents. In <italic toggle="yes">Proceedings of the 31st International Conference on Machine Learning (ICML-14)</italic>, pp. <fpage>1188</fpage>–<lpage>1196</lpage>.</mixed-citation></ref><ref id="bty263-B18"><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Luo</surname><given-names>Y.</given-names></name></person-group>
<etal>et al</etal> (<year>2017</year>) 
<article-title>Bridging semantics and syntax with graph algorithmsstate-of-the-art of extracting biomedical relations</article-title>. <source>Brief. Bioinformatics</source>, <volume>18</volume>, <fpage>160</fpage>–<lpage>178</lpage>.<pub-id pub-id-type="pmid">26851224</pub-id><pub-id pub-id-type="doi" assigning-authority="pmc">10.1093/bib/bbw001</pub-id><pub-id pub-id-type="pmcid">PMC5221425</pub-id></mixed-citation></ref><ref id="bty263-B19"><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Michal</surname><given-names>P.</given-names></name></person-group>
<etal>et al</etal> (<year>2011</year>) 
<article-title>Language combinatorics: a sentence pattern extraction architecture based on combinatorial explosion</article-title>. <source>Int. J. Comput. Linguistics</source>, <volume>2</volume>, <fpage>24</fpage>–<lpage>36</lpage>.</mixed-citation></ref><ref id="bty263-B20"><mixed-citation publication-type="other">
<person-group person-group-type="author"><name name-style="western"><surname>Mikolov</surname><given-names>T.</given-names></name></person-group>
<italic toggle="yes">et al.</italic> (<year>2013</year>) Distributed representations of words and phrases and their compositionality. In <italic toggle="yes">Advances in Neural Information Processing Systems</italic>, pp. <fpage>3111</fpage>–<lpage>3119</lpage>.</mixed-citation></ref><ref id="bty263-B21"><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Müller</surname><given-names>H.-M.</given-names></name></person-group>
<etal>et al</etal> (<year>2004</year>) 
<article-title>Textpresso: an ontology-based information retrieval and extraction system for biological literature</article-title>. <source>PLoS Biol</source>., <volume>2</volume>, <fpage>e309.</fpage><pub-id pub-id-type="pmid">15383839</pub-id><pub-id pub-id-type="doi" assigning-authority="pmc">10.1371/journal.pbio.0020309</pub-id><pub-id pub-id-type="pmcid">PMC517822</pub-id></mixed-citation></ref><ref id="bty263-B22"><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Nickel</surname><given-names>M.</given-names></name></person-group>
<etal>et al</etal> (<year>2016</year>) 
<article-title>A review of relational machine learning for knowledge graphs</article-title>. <source>Proc. IEEE</source>, <volume>104</volume>, <fpage>11</fpage>–<lpage>33</lpage>.</mixed-citation></ref><ref id="bty263-B23"><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Özgür</surname><given-names>A.</given-names></name></person-group>
<etal>et al</etal> (<year>2008</year>) 
<article-title>Identifying gene-disease associations using centrality on a literature mined gene-interaction network</article-title>. <source>Bioinformatics</source>, <volume>24</volume>, <fpage>i277</fpage>–<lpage>i285</lpage>.<pub-id pub-id-type="pmid">18586725</pub-id><pub-id pub-id-type="doi" assigning-authority="pmc">10.1093/bioinformatics/btn182</pub-id><pub-id pub-id-type="pmcid">PMC2718658</pub-id></mixed-citation></ref><ref id="bty263-B24"><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Papanikolaou</surname><given-names>N.</given-names></name></person-group>
<etal>et al</etal> (<year>2015</year>) 
<article-title>Protein–protein interaction predictions using text mining methods</article-title>. <source>Methods</source>, <volume>74</volume>, <fpage>47</fpage>–<lpage>53</lpage>.<pub-id pub-id-type="pmid">25448298</pub-id><pub-id pub-id-type="doi" assigning-authority="pmc">10.1016/j.ymeth.2014.10.026</pub-id></mixed-citation></ref><ref id="bty263-B25"><mixed-citation publication-type="other">
<person-group person-group-type="author"><name name-style="western"><surname>Rindflesch</surname><given-names>T.C.</given-names></name></person-group>
<etal>et al</etal> (<year>1999</year>) Edgar: extraction of drugs, genes and relations from the biomedical literature. In <italic toggle="yes">Biocomputing 2000</italic>, pp. <fpage>517</fpage>–<lpage>528</lpage>. World Scientific.<pub-id pub-id-type="doi" assigning-authority="pmc">10.1142/9789814447331_0049</pub-id><pub-id pub-id-type="pmcid">PMC2709525</pub-id><pub-id pub-id-type="pmid">10902199</pub-id></mixed-citation></ref><ref id="bty263-B26"><mixed-citation publication-type="other">
<person-group person-group-type="author"><name name-style="western"><surname>Schmitz</surname><given-names>M.</given-names></name></person-group>
<etal>et al</etal> (<year>2012</year>) Open language learning for information extraction. In: <italic toggle="yes">Proceedings of the 2012 Joint Conference on Empirical Methods in Natural Language Processing and Computational Natural Language Learning</italic>, pp. <fpage>523</fpage>–<lpage>534</lpage>. Association for Computational Linguistics.</mixed-citation></ref><ref id="bty263-B27"><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Segura-Bedmar</surname><given-names>I.</given-names></name></person-group>
<etal>et al</etal> (<year>2008</year>) 
<article-title>Drug name recognition and classification in biomedical texts: a case study outlining approaches underpinning automated systems</article-title>. <source>Drug Discov. Today</source>, <volume>13</volume>, <fpage>816</fpage>–<lpage>823</lpage>.<pub-id pub-id-type="pmid">18602492</pub-id><pub-id pub-id-type="doi" assigning-authority="pmc">10.1016/j.drudis.2008.06.001</pub-id></mixed-citation></ref><ref id="bty263-B28"><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Segura-Bedmar</surname><given-names>I.</given-names></name></person-group>
<etal>et al</etal> (<year>2011</year>) 
<article-title>The 1st DDIExtraction-2011 challenge task: extraction of drug-drug interactions from biomedical texts</article-title>. <source>CEUR workshop proc</source>, <volume>761</volume>, <fpage>1</fpage>–<lpage>9</lpage>.</mixed-citation></ref><ref id="bty263-B29"><mixed-citation publication-type="other">
<person-group person-group-type="author"><name name-style="western"><surname>Segura Bedmar</surname><given-names>I.</given-names></name></person-group>
<etal>et al</etal> (<year>2013</year>) Semeval-2013 task 9: extraction of drug-drug interactions from biomedical texts (ddiextraction 2013). In <italic toggle="yes">Second Joint Conference on Lexical and Computational Semantics (* SEM), Proceedings of the Seventh International Workshop on Semantic Evaluation (SemEval 2013)</italic>, Vol. 2, pp. 341–350.</mixed-citation></ref><ref id="bty263-B30"><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Seren</surname><given-names>Ü.</given-names></name></person-group>
<etal>et al</etal> (<year>2017</year>) 
<article-title>Arapheno: a public database for Arabidopsis thaliana phenotypes</article-title>. <source>Nucleic Acids Res</source>., <volume>45</volume>, <fpage>D1054</fpage>–<lpage>D1059</lpage>.<pub-id pub-id-type="pmid">27924043</pub-id><pub-id pub-id-type="doi" assigning-authority="pmc">10.1093/nar/gkw986</pub-id><pub-id pub-id-type="pmcid">PMC5210660</pub-id></mixed-citation></ref><ref id="bty263-B31"><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Singhal</surname><given-names>A.</given-names></name></person-group>
<etal>et al</etal> (<year>2016</year>) 
<article-title>Text mining genotype-phenotype relationships from biomedical literature for database curation and precision medicine</article-title>. <source>PLoS Comput. Biol</source>., <volume>12</volume>, <fpage>e1005017.</fpage><pub-id pub-id-type="pmid">27902695</pub-id><pub-id pub-id-type="doi" assigning-authority="pmc">10.1371/journal.pcbi.1005017</pub-id><pub-id pub-id-type="pmcid">PMC5130168</pub-id></mixed-citation></ref><ref id="bty263-B32"><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Tariq</surname><given-names>A.</given-names></name></person-group>
<etal>et al</etal> (<year>2017</year>) 
<article-title>Nelasso: group-sparse modeling for characterizing relations among named entities in news articles</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell</source>., <volume>39</volume>, <fpage>2000</fpage>–<lpage>2014</lpage>.<pub-id pub-id-type="pmid">27893385</pub-id><pub-id pub-id-type="doi" assigning-authority="pmc">10.1109/TPAMI.2016.2632117</pub-id></mixed-citation></ref><ref id="bty263-B33"><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Wei</surname><given-names>C.-H.</given-names></name></person-group>
<etal>et al</etal> (<year>2015</year>) 
<article-title>Gnormplus: an integrative approach for tagging genes, gene families, and protein domains</article-title>. <source>BioMed Res. Int</source>., <volume>2015</volume>, <fpage>1.</fpage><pub-id pub-id-type="doi" assigning-authority="pmc">10.1155/2015/918710</pub-id><pub-id pub-id-type="pmcid">PMC4561873</pub-id><pub-id pub-id-type="pmid">26380306</pub-id></mixed-citation></ref><ref id="bty263-B34"><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Wei</surname><given-names>C.-H.</given-names></name>, <name name-style="western"><surname>Kao</surname><given-names>H.-Y.</given-names></name></person-group> (<year>2011</year>) 
<article-title>Cross-species gene normalization by species inference</article-title>. <source>BMC Bioinformatics</source>, <volume>12</volume>, <fpage>S5.</fpage><pub-id pub-id-type="doi" assigning-authority="pmc">10.1186/1471-2105-12-S8-S5</pub-id><pub-id pub-id-type="pmcid">PMC3269940</pub-id><pub-id pub-id-type="pmid">22151999</pub-id></mixed-citation></ref><ref id="bty263-B35"><mixed-citation publication-type="other">
<person-group person-group-type="author"><name name-style="western"><surname>Xing</surname><given-names>W.</given-names></name></person-group>
<etal>et al</etal> (<year>2017</year>) Cascade word embedding to sentence embedding: A class label enhanced approach to phenotype extraction. In: <italic toggle="yes">2017 IEEE International Conference on Bioinformatics and Biomedicine (BIBM)</italic>, pp. <fpage>477</fpage>–<lpage>484</lpage>. IEEE.</mixed-citation></ref><ref id="bty263-B36"><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Xu</surname><given-names>Y.</given-names></name></person-group>
<etal>et al</etal> (<year>2009</year>) 
<article-title>MBA: a literature mining system for extracting biomedical abbreviations</article-title>. <source>BMC Bioinformatics</source>, <volume>10</volume>, <fpage>14.</fpage><pub-id pub-id-type="pmid">19134199</pub-id><pub-id pub-id-type="doi" assigning-authority="pmc">10.1186/1471-2105-10-14</pub-id><pub-id pub-id-type="pmcid">PMC2639376</pub-id></mixed-citation></ref><ref id="bty263-B37"><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>H.</given-names></name></person-group>
<etal>et al</etal> (<year>2015</year>) 
<article-title>Phenolyzer: phenotype-based prioritization of candidate genes for human diseases</article-title>. <source>Nat. Methods</source>, <volume>12</volume>, <fpage>841</fpage>–<lpage>843</lpage>.<pub-id pub-id-type="pmid">26192085</pub-id><pub-id pub-id-type="doi" assigning-authority="pmc">10.1038/nmeth.3484</pub-id><pub-id pub-id-type="pmcid">PMC4718403</pub-id></mixed-citation></ref><ref id="bty263-B38"><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Yang</surname><given-names>Z.</given-names></name></person-group>
<etal>et al</etal> (<year>2011</year>) 
<article-title>Multiple kernel learning in protein–protein interaction extraction from biomedical literature</article-title>. <source>Artif. Intell. Med</source>., <volume>51</volume>, <fpage>163</fpage>–<lpage>173</lpage>.<pub-id pub-id-type="pmid">21208788</pub-id><pub-id pub-id-type="doi" assigning-authority="pmc">10.1016/j.artmed.2010.12.002</pub-id></mixed-citation></ref><ref id="bty263-B39"><mixed-citation publication-type="other">
<person-group person-group-type="author"><name name-style="western"><surname>Zhu</surname><given-names>F.</given-names></name></person-group>
<etal>et al</etal> (<year>2015</year>) Protein-protein interaction network constructing based on text mining and reinforcement learning with application to prostate cancer. In <italic toggle="yes">Trustcom/BigDataSE/ISPA, 2015 IEEE</italic>, Vol. 1, pp. <fpage>1306</fpage>–<lpage>1311</lpage>. IEEE.<pub-id pub-id-type="doi" assigning-authority="pmc">10.1049/iet-syb.2014.0050</pub-id><pub-id pub-id-type="pmcid">PMC8687258</pub-id><pub-id pub-id-type="pmid">26243825</pub-id></mixed-citation></ref></ref-list></back></article>