<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article SYSTEM "http://jats.nlm.nih.gov/archiving/1.2/JATS-archivearticle1.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="1.2" article-type="research-article" xml:lang="en"><?properties open_access?><front><journal-meta><journal-id journal-id-type="publisher-id">13326</journal-id><journal-title-group><journal-title>Journal of Biomedical Semantics</journal-title><abbrev-journal-title abbrev-type="publisher">J Biomed Semant</abbrev-journal-title></journal-title-group><issn pub-type="epub">2041-1480</issn><publisher><publisher-name>BioMed Central</publisher-name><publisher-loc>London</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">s13326-021-00249-x</article-id><article-id pub-id-type="manuscript">249</article-id><article-id pub-id-type="doi">10.1186/s13326-021-00249-x</article-id><article-categories><subj-group subj-group-type="heading"><subject>Research</subject></subj-group><subj-group subj-group-type="article-collection" specific-use="Regular"><subject>International Conference on Biomedical Ontologies Direct to Journal Track</subject></subj-group></article-categories><title-group><article-title xml:lang="en">Linking common human diseases to their phenotypes; development of a resource for human phenomics</article-title></title-group><contrib-group><contrib contrib-type="author" id="Au1"><name><surname>Kafkas</surname><given-names>Şenay</given-names></name><xref ref-type="aff" rid="Aff1">1</xref></contrib><contrib contrib-type="author" id="Au2"><name><surname>Althubaiti</surname><given-names>Sara</given-names></name><xref ref-type="aff" rid="Aff1">1</xref></contrib><contrib contrib-type="author" id="Au3"><name><surname>Gkoutos</surname><given-names>Georgios V.</given-names></name><xref ref-type="aff" rid="Aff2">2</xref><xref ref-type="aff" rid="Aff3">3</xref></contrib><contrib contrib-type="author" corresp="yes" id="Au4"><contrib-id contrib-id-type="orcid">http://orcid.org/0000-0001-8149-5890</contrib-id><name><surname>Hoehndorf</surname><given-names>Robert</given-names></name><xref ref-type="aff" rid="Aff1">1</xref><xref ref-type="corresp" rid="IDs1332602100249x_cor4">d</xref></contrib><contrib contrib-type="author" id="Au5"><name><surname>Schofield</surname><given-names>Paul N.</given-names></name><xref ref-type="aff" rid="Aff4">4</xref></contrib><aff id="Aff1"><label>1</label><institution-wrap><institution-id institution-id-type="GRID">grid.45672.32</institution-id><institution-id institution-id-type="ISNI">0000 0001 1926 5090</institution-id><institution content-type="org-name">Computational Bioscience Research Center (CBRC), Computer, Electrical, and Mathematical Sciences &amp; Engineering Division, King Abdullah University of Science and Technology</institution></institution-wrap><addr-line content-type="street">4700 KAUST</addr-line><addr-line content-type="postcode">23955</addr-line><addr-line content-type="city">Thuwal</addr-line><country country="SA">Saudi Arabia</country></aff><aff id="Aff2"><label>2</label><institution-wrap><institution-id institution-id-type="GRID">grid.507332.0</institution-id><institution content-type="org-name">Health Data Research UK, Midlands site</institution></institution-wrap><addr-line content-type="street">Edgbaston</addr-line><addr-line content-type="postcode">B15 2TT</addr-line><addr-line content-type="city">Birmingham</addr-line><country country="GB">United Kingdom</country></aff><aff id="Aff3"><label>3</label><institution-wrap><institution-id institution-id-type="GRID">grid.6572.6</institution-id><institution-id institution-id-type="ISNI">0000 0004 1936 7486</institution-id><institution content-type="org-name">Institute of Cancer and Genomic Sciences, University of Birmingham</institution></institution-wrap><addr-line content-type="street">Edgbaston</addr-line><addr-line content-type="postcode">B15 2TT</addr-line><addr-line content-type="city">Birmingham</addr-line><country country="GB">United Kingdom</country></aff><aff id="Aff4"><label>4</label><institution-wrap><institution-id institution-id-type="GRID">grid.5335.0</institution-id><institution-id institution-id-type="ISNI">0000000121885934</institution-id><institution content-type="org-name">Department of Physiology, Development &amp; Neuroscience, University of Cambridge</institution></institution-wrap><addr-line content-type="street">Downing Street</addr-line><addr-line content-type="postcode">CB2 3EG</addr-line><addr-line content-type="city">Cambridge</addr-line><country country="GB">United Kingdom</country></aff></contrib-group><author-notes><corresp id="IDs1332602100249x_cor4"><label>d</label><email>robert.hoehndorf@kaust.edu.sa</email></corresp></author-notes><pub-date date-type="pub" publication-format="electronic"><day>23</day><month>8</month><year>2021</year></pub-date><pub-date date-type="collection" publication-format="electronic"><month>12</month><year>2021</year></pub-date><volume>12</volume><issue seq="17">1</issue><elocation-id>17</elocation-id><history><date date-type="registration"><day>5</day><month>8</month><year>2021</year></date><date date-type="received"><day>18</day><month>5</month><year>2021</year></date><date date-type="accepted"><day>30</day><month>7</month><year>2021</year></date><date date-type="online"><day>23</day><month>8</month><year>2021</year></date></history><permissions><copyright-statement>© The Author(s) 2021</copyright-statement><copyright-year>2021</copyright-year><license license-type="open-access" xlink:href="http://creativecommons.org/licenses/by/4.0/"><license-p><bold>Open Access</bold> This article is licensed under a Creative Commons Attribution 4.0 International License, which permits use, sharing, adaptation, distribution and reproduction in any medium or format, as long as you give appropriate credit to the original author(s) and the source, provide a link to the Creative Commons licence, and indicate if changes were made. The images or other third party material in this article are included in the article’s Creative Commons licence, unless indicated otherwise in a credit line to the material. If material is not included in the article’s Creative Commons licence and your intended use is not permitted by statutory regulation or exceeds the permitted use, you will need to obtain permission directly from the copyright holder. To view a copy of this licence, visit <ext-link xlink:href="http://creativecommons.org/licenses/by/4.0/" ext-link-type="url">http://creativecommons.org/licenses/by/4.0/</ext-link>. The Creative Commons Public Domain Dedication waiver (<ext-link xlink:href="http://creativecommons.org/publicdomain/zero/1.0/" ext-link-type="url">http://creativecommons.org/publicdomain/zero/1.0/</ext-link>) applies to the data made available in this article, unless otherwise stated in a credit line to the data.</license-p></license></permissions><abstract id="Abs1" xml:lang="en"><title>Abstract</title><sec id="ASec1"><title>Background</title><p id="Par1">In recent years a large volume of clinical genomics data has become available due to rapid advances in sequencing technologies. Efficient exploitation of this genomics data requires linkage to patient phenotype profiles. Current resources providing disease-phenotype associations are not comprehensive, and they often do not have broad coverage of the disease terminologies, particularly ICD-10, which is still the primary terminology used in clinical settings.</p></sec><sec id="ASec2"><title>Methods</title><p id="Par2">We developed two approaches to gather disease-phenotype associations. First, we used a text mining method that utilizes semantic relations in phenotype ontologies, and applies statistical methods to extract associations between diseases in ICD-10 and phenotype ontology classes from the literature. Second, we developed a semi-automatic way to collect ICD-10–phenotype associations from existing resources containing known relationships.</p></sec><sec id="ASec3"><title>Results</title><p id="Par3">We generated four datasets. Two of them are independent datasets linking diseases to their phenotypes based on text mining and semi-automatic strategies. The remaining two datasets are generated from these datasets and cover a subset of ICD-10 classes of common diseases contained in UK Biobank. We extensively validated our text mined and semi-automatically curated datasets by: comparing them against an expert-curated validation dataset containing disease–phenotype associations, measuring their similarity to disease–phenotype associations found in public databases, and assessing how well they could be used to recover gene–disease associations using phenotype similarity.</p></sec><sec id="ASec4"><title>Conclusion</title><p id="Par4">We find that our text mining method can produce phenotype annotations of diseases that are correct but often too general to have significant information content, or too specific to accurately reflect the typical manifestations of the sporadic disease. On the other hand, the datasets generated from integrating multiple knowledgebases are more complete (i.e., cover more of the required phenotype annotations for a given disease). We make all data freely available at <ext-link xlink:href="https://doi.org/10.5281/zenodo.4726713" ext-link-type="doi">https://doi.org/10.5281/zenodo.4726713</ext-link>.</p></sec></abstract><kwd-group xml:lang="en"><title>Keywords</title><kwd>Disease–phenotype associations</kwd><kwd>Ontologies</kwd><kwd>Text mining</kwd><kwd>UK Biobank</kwd></kwd-group><funding-group><award-group><funding-source><institution-wrap><institution>King Abdullah University of Science and Technology</institution><institution-id institution-id-type="doi" vocab="open-funder-registry">http://dx.doi.org/10.13039/501100004052</institution-id></institution-wrap></funding-source><award-id award-type="FundRef grant">URF/1/3790-01-01</award-id></award-group><award-group><funding-source><institution-wrap><institution>King Abdullah University of Science and Technology</institution><institution-id institution-id-type="doi" vocab="open-funder-registry">http://dx.doi.org/10.13039/501100004052</institution-id></institution-wrap></funding-source><award-id award-type="FundRef grant">URF/1/4355-01-01</award-id></award-group><award-group><funding-source><institution-wrap><institution>king abdullah university of science and technology</institution><institution-id institution-id-type="doi" vocab="open-funder-registry">http://dx.doi.org/10.13039/501100004052</institution-id></institution-wrap></funding-source><award-id award-type="FundRef grant">FCC/1/1976-28-01</award-id></award-group><award-group><funding-source><institution-wrap><institution>king abdullah university of science and technology</institution><institution-id institution-id-type="doi" vocab="open-funder-registry">http://dx.doi.org/10.13039/501100004052</institution-id></institution-wrap></funding-source><award-id award-type="FundRef grant">FCC/1/1976-29-01</award-id></award-group></funding-group><custom-meta-group><custom-meta><meta-name>publisher-imprint-name</meta-name><meta-value>BioMed Central</meta-value></custom-meta><custom-meta><meta-name>volume-issue-count</meta-name><meta-value>1</meta-value></custom-meta><custom-meta><meta-name>issue-article-count</meta-name><meta-value>17</meta-value></custom-meta><custom-meta><meta-name>issue-toc-levels</meta-name><meta-value>0</meta-value></custom-meta><custom-meta><meta-name>issue-pricelist-year</meta-name><meta-value>2021</meta-value></custom-meta><custom-meta><meta-name>issue-copyright-holder</meta-name><meta-value>The Author(s)</meta-value></custom-meta><custom-meta><meta-name>issue-copyright-year</meta-name><meta-value>2021</meta-value></custom-meta><custom-meta><meta-name>article-contains-esm</meta-name><meta-value>Yes</meta-value></custom-meta><custom-meta><meta-name>article-numbering-style</meta-name><meta-value>Unnumbered</meta-value></custom-meta><custom-meta><meta-name>article-registration-date-year</meta-name><meta-value>2021</meta-value></custom-meta><custom-meta><meta-name>article-registration-date-month</meta-name><meta-value>8</meta-value></custom-meta><custom-meta><meta-name>article-registration-date-day</meta-name><meta-value>5</meta-value></custom-meta><custom-meta><meta-name>article-toc-levels</meta-name><meta-value>0</meta-value></custom-meta><custom-meta><meta-name>toc-levels</meta-name><meta-value>0</meta-value></custom-meta><custom-meta><meta-name>volume-type</meta-name><meta-value>Regular</meta-value></custom-meta><custom-meta><meta-name>journal-product</meta-name><meta-value>ArchiveJournal</meta-value></custom-meta><custom-meta><meta-name>numbering-style</meta-name><meta-value>Unnumbered</meta-value></custom-meta><custom-meta><meta-name>article-collection-editor</meta-name><meta-value>Janna Hastings;Robert Hoehndorf</meta-value></custom-meta><custom-meta><meta-name>article-grants-type</meta-name><meta-value>OpenChoice</meta-value></custom-meta><custom-meta><meta-name>metadata-grant</meta-name><meta-value>OpenAccess</meta-value></custom-meta><custom-meta><meta-name>abstract-grant</meta-name><meta-value>OpenAccess</meta-value></custom-meta><custom-meta><meta-name>bodypdf-grant</meta-name><meta-value>OpenAccess</meta-value></custom-meta><custom-meta><meta-name>bodyhtml-grant</meta-name><meta-value>OpenAccess</meta-value></custom-meta><custom-meta><meta-name>bibliography-grant</meta-name><meta-value>OpenAccess</meta-value></custom-meta><custom-meta><meta-name>esm-grant</meta-name><meta-value>OpenAccess</meta-value></custom-meta><custom-meta><meta-name>online-first</meta-name><meta-value>false</meta-value></custom-meta><custom-meta><meta-name>pdf-file-reference</meta-name><meta-value>BodyRef/PDF/13326_2021_Article_249.pdf</meta-value></custom-meta><custom-meta><meta-name>pdf-type</meta-name><meta-value>Typeset</meta-value></custom-meta><custom-meta><meta-name>target-type</meta-name><meta-value>OnlinePDF</meta-value></custom-meta><custom-meta><meta-name>issue-type</meta-name><meta-value>Regular</meta-value></custom-meta><custom-meta><meta-name>article-type</meta-name><meta-value>OriginalPaper</meta-value></custom-meta><custom-meta><meta-name>journal-subject-primary</meta-name><meta-value>Mathematics</meta-value></custom-meta><custom-meta><meta-name>journal-subject-secondary</meta-name><meta-value>Algorithms</meta-value></custom-meta><custom-meta><meta-name>journal-subject-secondary</meta-name><meta-value>Computer Appl. in Life Sciences</meta-value></custom-meta><custom-meta><meta-name>journal-subject-secondary</meta-name><meta-value>Data Mining and Knowledge Discovery</meta-value></custom-meta><custom-meta><meta-name>journal-subject-secondary</meta-name><meta-value>Computational Biology/Bioinformatics</meta-value></custom-meta><custom-meta><meta-name>journal-subject-secondary</meta-name><meta-value>Bioinformatics</meta-value></custom-meta><custom-meta><meta-name>journal-subject-secondary</meta-name><meta-value>Combinatorial Libraries</meta-value></custom-meta><custom-meta><meta-name>journal-subject-collection</meta-name><meta-value>Mathematics and Statistics</meta-value></custom-meta><custom-meta><meta-name>open-access</meta-name><meta-value>true</meta-value></custom-meta></custom-meta-group></article-meta><notes notes-type="ESMHint"><title>Supplementary Information</title><p>The online version contains supplementary material available at (<ext-link xlink:href="https://doi.org/10.1186/s13326-021-00249-x" ext-link-type="doi">https://doi.org/10.1186/s13326-021-00249-x</ext-link>).</p></notes></front><body><sec id="Sec1"><title>Background</title><p>The genomic revolution in medicine has been driven by the huge success of next generation sequencing and the availability of millions of genome or exome sequences [<xref ref-type="bibr" rid="CR1">1</xref>, <xref ref-type="bibr" rid="CR2">2</xref>]. However, the utility of genomic sequence is determined firstly by our knowledge of the relationship between genomic variants and disease conditions or predispositions [<xref ref-type="bibr" rid="CR3">3</xref>–<xref ref-type="bibr" rid="CR5">5</xref>], and secondly by our knowledge of the relationship between protein function and sequence [<xref ref-type="bibr" rid="CR6">6</xref>, <xref ref-type="bibr" rid="CR7">7</xref>]. To date, the application of this knowledge has been focused in two areas, rare diseases, generally Mendelian [<xref ref-type="bibr" rid="CR8">8</xref>], and common or complex diseases [<xref ref-type="bibr" rid="CR5">5</xref>, <xref ref-type="bibr" rid="CR9">9</xref>].</p><p>A significant breakthrough in rare disease diagnostics and candidate gene discovery was facilitated by the development of phenotype ontologies that capture the phenotypes associated with a disease entity. These are now available not only for human and most of the model organisms [<xref ref-type="bibr" rid="CR10">10</xref>, <xref ref-type="bibr" rid="CR11">11</xref>], but also as unified and integrated phenotype ontologies [<xref ref-type="bibr" rid="CR12">12</xref>, <xref ref-type="bibr" rid="CR13">13</xref>] where the equivalences and relationships between phenotypes in different species are captured. Use of these ontologies to establish phenotypic similarity between undiagnosed patients and known human disease entities, or model organism mutants, has provided useful new diagnostic support and discovery tools [<xref ref-type="bibr" rid="CR14">14</xref>–<xref ref-type="bibr" rid="CR16">16</xref>]. This work has benefited considerably from careful phenotypic characterization of the known rare recurrent and Mendelian diseases of which there are estimated to be about 7,000 [<xref ref-type="bibr" rid="CR10">10</xref>, <xref ref-type="bibr" rid="CR17">17</xref>, <xref ref-type="bibr" rid="CR18">18</xref>]. Efforts to annotate complex and common diseases with their phenotypes have been limited by the scale of the task, with 14,400 distinct diagnoses and causes of death identified in ICD-10 [<xref ref-type="bibr" rid="CR19">19</xref>] and 69,000 diagnosis codes in ICD-10-CM, of which many are phenotypic variants.</p><p>The advantage of creating a corpus of phenotype annotations to common diseases is that they can then be used to computationally search for phenotypic and genetic associations between phenotypes or diseases, for identification of new phenotypic subgroups, and for diagnostic support and facilitation of the analysis of electronic patient record data [<xref ref-type="bibr" rid="CR20">20</xref>–<xref ref-type="bibr" rid="CR23">23</xref>]. In recent years, evidence has accumulated to support a model in which different diseases have common underlying etiopathological mechanisms and shared phenotypes, or endophenotypes [<xref ref-type="bibr" rid="CR24">24</xref>, <xref ref-type="bibr" rid="CR25">25</xref>]. The hypothesis that similarity between phenotypes reflects underlying biological modules of functionally related genes has been convincingly demonstrated for Mendelian disease but little work has been done for common and complex disease [<xref ref-type="bibr" rid="CR26">26</xref>, <xref ref-type="bibr" rid="CR27">27</xref>]. An earlier study of common diseases [<xref ref-type="bibr" rid="CR28">28</xref>] suggested that these diseases form modules related to Mendelian genes with similar phenotypes. Similarly, the phenotype study by Ghiassian et al. [<xref ref-type="bibr" rid="CR24">24</xref>] took three selected endophenotypes (inflammation, thrombosis, and fibrosis) and genes annotated to these, to show that the genetic modules associated with each phenotype interacted together to generate inflammation.</p><p>We have previously used text mining to annotate the diseases covered by the Human Disease Ontology (DO) [<xref ref-type="bibr" rid="CR29">29</xref>], which contains both common and rare diseases, and demonstrated that phenotypically closely related diseases are linked at the level of underlying etiology and align with existing nosology [<xref ref-type="bibr" rid="CR30">30</xref>]. However, this set of annotations was limited at the time to those diseases in DO and used only literature-derived associations. Many disease-phenotype pairs have been gathered from human genetic studies and animal model experiments and are now available from large-scale public resources [<xref ref-type="bibr" rid="CR31">31</xref>, <xref ref-type="bibr" rid="CR32">32</xref>]; one aim of the current study was to leverage these curated public resources alongside a more comprehensive text mining effort. However, these resources are far from being complete, and few phenotypes are linked to terms found in ICD-10 [<xref ref-type="bibr" rid="CR33">33</xref>], which is the primary disease terminology used in clinical practice. A large number of disease–phenotype associations are still latent in the literature and require automated methods to extract.</p><p>Here, we focus on the diseases in ICD-10 and link them to their relevant phenotypes from the Human Phenotype (HPO) and Mammalian Phenotype ontologies (MP) [<xref ref-type="bibr" rid="CR34">34</xref>, <xref ref-type="bibr" rid="CR35">35</xref>]. We present two approaches to gather disease-phenotype associations. One of them is text mining from the literature and the other one is semi-automatic harvesting from publicly available curated resources. To extract the disease-phenotype associations from text, we utilize the semantics of the PhenomeNET ontology to increase the coverage of annotations that are not explicitly mentioned [<xref ref-type="bibr" rid="CR12">12</xref>] in text, and apply a statistical approach to find significant associations between a disease and sets of phenotypes. We evaluate our text mining predictions against the known disease–phenotype associations from the HPO database [<xref ref-type="bibr" rid="CR31">31</xref>]. Furthermore, we demonstrate the utility of the generated datasets in predicting gene–disease associations from Mouse Genome Informatics (MGI) [<xref ref-type="bibr" rid="CR32">32</xref>] based on phenotype similarity. We provide all of the datasets of disease–phenotype associations at <ext-link xlink:href="https://doi.org/10.5281/zenodo.4726713" ext-link-type="doi">https://doi.org/10.5281/zenodo.4726713</ext-link>.</p></sec><sec id="Sec2" sec-type="materials|methods"><title>Materials and methods</title><sec id="Sec3"><title>Resources used</title><sec id="Sec4"><title>Diseases</title><p>We have built a semantic disease resource from the Unified Medical Language System (UMLS) [<xref ref-type="bibr" rid="CR36">36</xref>] and use it for disease–phenotype association extraction. We use UMLS due to its completeness, but also its extensively validated mappings to other terminologies, specifically ICD-10 and HPO, which are not available elsewhere. As one of our main aims is to facilitate the use of data in Electronic Health Records, we focus here on diseases included in the ICD-10. However, it is straightforward to map our resource onto other known disease resources such as DO [<xref ref-type="bibr" rid="CR29">29</xref>] and Mondo Disease Ontology [<xref ref-type="bibr" rid="CR37">37</xref>] through the ICD mappings in them.</p><p>To generate our disease resource, we first parsed UMLS data (from the file MRCONSO.RRF in UMLS, downloaded on 04/11/2019) and gathered all the disease concepts along with their labels, definitions and sub-class relations which were mapped to any of the classes in ICD-10, SNOMED CT, HPO, or Online Mendelian Inheritance in Man (OMIM) [<xref ref-type="bibr" rid="CR38">38</xref>]. We then represent the resulting integrated resources using the Web Ontology Language (OWL) [<xref ref-type="bibr" rid="CR39">39</xref>] where we assert rdfs:subClassOf between two classes that are sub-classes in UMLS, and rdfs:label properties based on all the collected labels and definitions. Although our main focus was ICD-10 diseases, we included disease classes represented in SNOMED, HPO, and OMIM as well in the generation of this resource so as to benefit from their ontological structure and maintain an asserted hierarchy among the disease concepts. This semantic disease resource covers a total of 1,535,927 disease labels from 519,735 disease concepts. We used this data to identify disease mentions (names, synonyms and acronyms) in text.</p></sec><sec id="Sec5"><title>Phenotypes</title><p>We used two phenotype ontologies, HPO and MP, to identify phenotypes in text. Both ontologies contain classes that are relevant to humans, so to cover the complete phenotype profile of a given disease as completely as possible we used MP in addition to HPO. We used only the subclasses of the <italic>Phenotypic abnormality</italic> branch of HPO (14,749 HPO classes in total)to generate the text-mined dataset as this branch covers phenotypes that can be readily associated with diseases. For the semi-automatically generated dataset, we considered all of the HPO classes as the data is seeded from curated annotations. We used primary and alternative class labels along with synonyms in text matching. We used the PhenomeNET [<xref ref-type="bibr" rid="CR40">40</xref>] ontology, which includes the phenotypes from HPO and MP, to generate embeddings for diseases and genes. Briefly, PhenomeNET is developed by transforming phenotype ontologies into a formal representation, combining phenotype ontologies with anatomy ontologies, and applying a measure of semantic similarity to construct a cross-species phenotype network.</p></sec><sec id="Sec6"><title>Known disease-phenotype, gene-phenotype and gene-disease associations</title><p>We gathered the known disease–phenotype associations from UMLS [<xref ref-type="bibr" rid="CR36">36</xref>], the HPO database [<xref ref-type="bibr" rid="CR31">31</xref>] on 12/10/2020 and Wikidata [<xref ref-type="bibr" rid="CR41">41</xref>] on 13/09/2020. We gathered the known mouse gene–phenotype [<xref ref-type="bibr" rid="CR42">42</xref>] as well as gene–disease associations [<xref ref-type="bibr" rid="CR43">43</xref>] from MGI [<xref ref-type="bibr" rid="CR32">32</xref>] on 15/03/2021.</p></sec><sec id="Sec7"><title>UK biobank</title><p>We generated a list of 2,106 common diseases from the UK Biobank identified by their ICD-10 codes. To generate this list, we considered only ICD-10 codes that have 100 or more patients in UK Biobank, as identified through the main or secondary diagnosis fields.</p></sec></sec><sec id="Sec8"><title>Generating disease-phenotype associations</title><p>In this study, we developed two methods; one of them is a text-mining based method and the other one is a semi-automatic way to gather asserted disease–phenotype associations from public data resources. By using these two methods, we generated a total of four datasets in this study. Figure <xref rid="Fig1" ref-type="fig">1</xref> depicts an overview of the processes applied and the datasets generated. First, we generated two independent datasets; one by text mining disease–phenotype associations from PubMed abstracts (this dataset is labeled “Text Mined”) and an another one by semi-automatically gathering the known disease-phenotype associations from multiple resources (labeled “Semi-automatic”).
<fig id="Fig1"><label>Fig. 1</label><caption xml:lang="en"><p>Overview of generating disease-phenotype associations. We used two methods, text mining and a semi-automatic way to generate two independent datasets. We further generated subsets of these datasets covering the phenotype associations of common diseases from UK Biobank. To generate the semi-automatic (UKB) dataset, we manually curated the selected diseases</p></caption><p><graphic specific-use="HTML" mime-subtype="PNG" xlink:href="MediaObjects/13326_2021_249_Fig1_HTML.png" id="MO1"/></p></fig></p><p>Linking genotype and phenotype can be extremely helpful in DNA sequence analysis for revealing causative variants. Therefore, we selected the disease–phenotype associations linked to common diseases in the UK Biobank from these two datasets. In the case of the Text Mined dataset, we retrieved all of the selected associations into the data subset we call “Text Mined (UKB)”. In the case of the Semi-automatic dataset, we further applied expert manual curation on the selected associations and generated the “Semi-automatic (UKB)” dataset.</p><sec id="Sec9"><title>Text mining disease-phenotype associations</title><p>Figure <xref rid="Fig2" ref-type="fig">2</xref> depicts the overview of text mining of disease-phenotype associations. To extract disease–phenotype associations, we first indexed approximately 30 million PubMed abstracts downloaded on 22/09/2019 from https://ftp://ftp.ncbi.nlm.nih.gov/pubmed/baseline using Apache Lucene [<xref ref-type="bibr" rid="CR44">44</xref>]. Secondly, we identified the abstract level occurrence and co-occurrence of each disease–phenotype pair. We then propagated the co-occurrence statistics by using the semantics in PhenomeNET (i.e., if <italic>C</italic> is a subclass of <italic>D</italic> in the PhenomeNET ontology, every mention of <italic>C</italic> in an abstract is also considered a mention of <italic>D</italic>). and calculated the normalized pointwise mutual information (NPMI) [<xref ref-type="bibr" rid="CR45">45</xref>] to measure the strength of the association.
<fig id="Fig2"><label>Fig. 2</label><caption xml:lang="en"><p>Overview of text mining disease-phenotype associations. Disease labels are gathered from OMIM, SNOMED and HPO records which are linked to ICD-10 in UMLS. Phenotype labels are gathered from HPO and MP</p></caption><p><graphic specific-use="HTML" mime-subtype="PNG" xlink:href="MediaObjects/13326_2021_249_Fig2_HTML.png" id="MO2"/></p></fig></p><p>NPMI is a measure of collocation of two terms. As the disease and phenotype concepts are represented by ontology classes, we reformulated the NPMI to measure the collocation between two classes. First, we identify the set of labels and synonyms associated with every class; <italic>L</italic><italic>a</italic><italic>b</italic><italic>e</italic><italic>l</italic><italic>s</italic>(<italic>C</italic>) denotes the set of labels and synonyms of <italic>C</italic>. We then define <italic>T</italic><italic>e</italic><italic>r</italic><italic>m</italic><italic>s</italic>(<italic>C</italic>) as the set of all terms that can be used to refer to <italic>C</italic>: <inline-formula id="IEq1"><alternatives><mml:math><mml:mtext mathvariant="italic">Terms</mml:mtext><mml:mo>(</mml:mo><mml:mi>C</mml:mi><mml:mo>)</mml:mo><mml:mo>:</mml:mo><mml:mo>=</mml:mo><mml:mo>{</mml:mo><mml:mi>x</mml:mi><mml:mo>|</mml:mo><mml:mi>x</mml:mi><mml:mo>∈</mml:mo><mml:mtext mathvariant="italic">Labels</mml:mtext><mml:mo>(</mml:mo><mml:mi>S</mml:mi><mml:mo>)</mml:mo><mml:mo>∧</mml:mo><mml:mi>S</mml:mi><mml:mo>⊑</mml:mo><mml:mi>C</mml:mi><mml:mo>}</mml:mo></mml:math><tex-math id="IEq1_TeX">\documentclass[12pt]{minimal}
				\usepackage{amsmath}
				\usepackage{wasysym}
				\usepackage{amsfonts}
				\usepackage{amssymb}
				\usepackage{amsbsy}
				\usepackage{mathrsfs}
				\usepackage{upgreek}
				\setlength{\oddsidemargin}{-69pt}
				\begin{document}$Terms(C) := \{ x | x \in Labels(S) \land S \sqsubseteq C \}$\end{document}</tex-math><inline-graphic xlink:href="13326_2021_249_Article_IEq1.gif"/></alternatives></inline-formula>.</p><p>We calculated the NPMI between classes <italic>C</italic> and <italic>D</italic> as 
<disp-formula id="Equ1"><label>1</label><alternatives><mml:math><mml:mtext mathvariant="italic">npmi</mml:mtext><mml:mo>(</mml:mo><mml:mi>C</mml:mi><mml:mo>,</mml:mo><mml:mi>D</mml:mi><mml:mo>)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mo>log</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mi>C</mml:mi><mml:mo>,</mml:mo><mml:mi>D</mml:mi></mml:mrow></mml:msub><mml:mo>·</mml:mo><mml:msub><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mtext mathvariant="italic">tot</mml:mtext></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mi>C</mml:mi></mml:mrow></mml:msub><mml:mo>·</mml:mo><mml:msub><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mi>D</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mrow><mml:mrow><mml:mo>−</mml:mo><mml:mo>log</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mi>C</mml:mi><mml:mo>,</mml:mo><mml:mi>D</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mtext mathvariant="italic">tot</mml:mtext></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mrow></mml:mfrac></mml:math><tex-math id="Equ1_TeX">\documentclass[12pt]{minimal}
				\usepackage{amsmath}
				\usepackage{wasysym}
				\usepackage{amsfonts}
				\usepackage{amssymb}
				\usepackage{amsbsy}
				\usepackage{mathrsfs}
				\usepackage{upgreek}
				\setlength{\oddsidemargin}{-69pt}
				\begin{document} $$ npmi(C,D) = \frac{\log{\frac{n_{C,D}\cdot n_{tot}}{n_{C} \cdot n_{D}}}}{-\log{\frac{n_{C,D}}{n_{tot}}}}  $$ \end{document}</tex-math><graphic position="anchor" xlink:href="13326_2021_249_Article_Equ1.gif"/></alternatives></disp-formula></p><p>where <italic>n</italic><sub><italic>tot</italic></sub> is the total number of abstracts in our corpus in which at least one disease and one phenotype name co-occur, <italic>n</italic><sub><italic>C</italic>,<italic>D</italic></sub> is the number of abstracts in which both a term from <italic>T</italic><italic>e</italic><italic>r</italic><italic>m</italic><italic>s</italic>(<italic>C</italic>) and a term from <italic>T</italic><italic>e</italic><italic>r</italic><italic>m</italic><italic>s</italic>(<italic>D</italic>) co-occur, <italic>n</italic><sub><italic>C</italic></sub> is the number of abstracts in which a term from <italic>T</italic><italic>e</italic><italic>r</italic><italic>m</italic><italic>s</italic>(<italic>C</italic>) occurs, and <italic>n</italic><sub><italic>D</italic></sub> is the number of abstracts in which a term from <italic>T</italic><italic>e</italic><italic>r</italic><italic>m</italic><italic>s</italic>(<italic>D</italic>) occurs.</p></sec><sec id="Sec10"><title>Gathering disease-phenotype associations from public data sources</title><p>We generated the Semi-automatic dataset of disease–phenotype associations based on gathering known associations from UMLS, Wikidata, and the HPO database, and then propagating them based on the superclass relations defined in ICD-10 and lexical match of superclasses of diseases in the HPO dataset (see Fig. <xref rid="Fig3" ref-type="fig">3</xref>). To generate this dataset, first, we gathered the known disease–phenotype associations from UMLS and Wikidata. We collected a total number of 2,340 ICD-10–HPO direct mappings for 2,268 distinct diseases from UMLS. We used SPARQL queries and retrieved a total number of 2,029 distinct associations for 404 distinct diseases (mapped to ICD-10) with their symptoms and secondary effects (mapped to HPO) from Wikidata; the SPARQL queries we used are available as Supplementary Materials. Second, for capturing the known disease–phenotype associations from the HPO database, we applied an additional disease identifier conversion step. The HPO database contains disease–phenotype associations where the diseases are mapped to OMIM and the phenotypes are mapped to HPO. To include these associations, we mapped the OMIM diseases to their ICD-10 codes by using the ICD-10–OMIM mappings from UMLS and Wikidata. We gathered the mappings from Wikidata using another SPARQL query (see). We extracted 303 and 5,747 ICD-10–OMIM mappings from UMLS and Wikidata, respectively. Merging these two resources, we obtained a total of 5,845 distinct ICD-10–OMIM mappings where 1,447 distinct ICD-10 codes are mapped to their corresponding OMIM identifiers. Altogether, utilizing the obtained ICD-10–OMIM mappings, we gathered a total number of 41,529 ICD-10–HPO associations for 1,366 distinct diseases from the HPO database. Third, we filtered out the associations involving 21 generic, not informative phenotypes which were manually identified by an expert from the dataset (e.g., HP:0000005<italic>Mode of inheritance</italic>, HP:0000006<italic>Autosomal dominant inheritance</italic>, HP:0012824<italic>Severity</italic>, HP:0025285<italic>Aggravated by</italic>, HP:0012834 Right). Fourth, we propagated the known annotations based on the superclass relations in ICD-10 coding system hierarchy. For example, phenotypes linked to (ICD-10:G30) <italic>Alzheimer’s disease</italic> are propagated to all 4 of its sub-classes (ICD-10:G30.0 Alzheimer’s disease with early onset, ICD-10:G30.1 Alzheimer’s disease with late onset, ICD-10:G30.8<italic>Other Alzheimer’s disease</italic>, ICD-10:G30.9<italic>Alzheimer’s disease, unspecified</italic>). We also propagated annotations to the diseases from their superclasses that we find by lexical match in the HPO database. For example, we linked ICD-10:I84.4, <italic>External hemorrhoids with complications</italic> to <italic>Hemorrhoids</italic> (HP:0032551) (see Fig. <xref rid="Fig3" ref-type="fig">3</xref>c).
<fig id="Fig3"><label>Fig. 3</label><caption xml:lang="en"><p>Overview of semi-automatic gathering known disease-phenotype associations. (a) Collecting known ICD-10-HPO associations from WikiData, UMLS and HPO, filtering associations with generic phenotypes and propagation based on ICD-10 hierarchy and lexical match of disease super-classes in HPO. (b) A sample ICD-10 Hierarchy. (c) A sample lexical match of disease super-class in HPO</p></caption><p><graphic specific-use="HTML" mime-subtype="PNG" xlink:href="MediaObjects/13326_2021_249_Fig3_HTML.png" id="MO3"/></p></fig></p></sec><sec id="Sec11"><title>Manual curation of “Semi-automatic” disease–phenotype associations</title><p>Linking genotype and phenotype is important for understanding the underlying mechanisms of genomic disorders. Therefore, we selected the phenotype associations of the common diseases in UK Biobank from this dataset and manually curated them. We named this curated dataset “Semi-automatic (UKB)” and released it as an additional resource. To generate this subset of data, we first identified and prioritized a total of 2,106 diseases which are common, defined as those for which at least 100 individuals in UK Biobank have either a primary or secondary diagnosis. Second, we retrieved the known and propagated associations of these 2,106 diseases. Third, we manually added the phenotypes for the diseases for which we could not find any associations after applying the second step. Missing ICD-10–HPO annotations were provided by expert curation and reference to the literature.</p><p>Clinical presentation was checked initially using expert knowledge supported by standard texts [<xref ref-type="bibr" rid="CR46">46</xref>, <xref ref-type="bibr" rid="CR47">47</xref>] and then examined in further depth using recent literature reviews and papers. Examination of the HPO classes annotated to diseases in this semi-automated dataset revealed several broad curation strategies for annotation taken by each contributing dataset; UMLS, Wikidata, and HPO database. Each type of annotation was not limited to one data source (for example direct mapping was found in all contributing datasets), and reflects the various strategies and individual pragmatic decisions adopted by the contributing resources. This is a compromise in comparison with a priori expert annotation, but our overall validation suggests that it does not compromise the utility of our datasets. Several examples are shown below: 
<list list-type="order"><list-item><p><italic>Direct mapping</italic> In some cases, HPO classes reflect a simple mapping from ICD-10 to HPO; for example ICD-10:N20.0<italic>Calculus of kidney</italic> is annotated with HP:0000121 Nephrocalcinosis.</p></list-item><list-item><p><italic>High level mapping</italic>ICD-10:H40.1, Primary open-angle glaucoma is annotated with HP:0000478, <italic>Abnormality of the eye</italic>.</p></list-item><list-item><p><italic>Symptom manifestation</italic>ICD-10:C91.1, Chronic lymphocytic leukaemia is annotated with HP:0040088, <italic>Abnormal lymphocyte count</italic></p></list-item><list-item><p><italic>Associated phenotypes including etiological predication, and closely related diseases or phenotypes</italic>ICD-10:E21.0, <italic>Primary hyperparathyroidism</italic> is annotated with HP:0011769, <italic>Ectopic parathyroid</italic>; similarly, ICD-10:D75.2, <italic>Essential thrombocytosis</italic>, is annotated with HP:0011974, <italic>Myelofibrosis</italic>, a closely linked disorder [<xref ref-type="bibr" rid="CR48">48</xref>].</p></list-item></list></p><p>Each method of assigning annotations used by the contributing resources had, to a greater or lesser degree, a bias towards one or more of these patterns (see below). While restricting annotations simply to signs and symptoms would have met the aim of disaggregating diseases into their constituent phenotypes, much information would have been lost and in fact HPO itself includes annotations of all of these types to advantage. We show below that using all of the methods of annotation resulted in a much better predictive outcome on the disease-gene validation task, justifying the inclusion of all of these types in annotation. The possible bias introduced by annotation density and the level of annotation to superclasses is discussed below.</p><p>In the last step, the generated ICD-10–HPO associations for 2106 ICD-10 codes were assessed by a biomedical expert for inappropriate and incorrect HPO annotations (false positives) and removed. The final number of common ICD-10 codes that could be linked to their phenotypes is 1,995.</p></sec></sec><sec id="Sec12"><title>Measuring phenotypic similarities of genes and diseases</title><p>We used OWL2Vec* [<xref ref-type="bibr" rid="CR49">49</xref>] to generate embeddings for entities based on their associations with phenotypes. OWL2Vec* is a method to generate embeddings for classes in OWL ontologies. OWL2Vec* converts an ontology into a graph based on syntactic patterns represented in the ontology axioms; in the graph, nodes correspond to either classes or individuals in the ontology, and edges correspond to axiom patterns. OWL2Vec* then applies a graph embedding method based on random walks to nodes in this graph; it explores the node neighborhood through iterated random walks with a subtree kernel and uses Word2Vec [<xref ref-type="bibr" rid="CR50">50</xref>] to embed nodes within a vector space.</p><p>For each disease, we generated two ontology embeddings, one based on the text mined phenotype profile and the other on the phenotype profile from the HPO database. We used the default parameter settings of the OWL2Vec* implementation: vector size 100, window size 5, minimum occurrence count of 1, skip-gram (sg) model, random walk with depth (number of walk) of 3.</p><p>We measured the similarity between the ontology embeddings using cosine similarity: 
<disp-formula id="Equ2"><label>2</label><alternatives><mml:math><mml:mtext>sim</mml:mtext><mml:mfenced close=")" open="("><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:mfenced><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>·</mml:mo><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mfenced close="∥" open="∥"><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:mfenced><mml:mfenced close="∥" open="∥"><mml:mrow><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:mfenced></mml:mrow></mml:mfrac></mml:math><tex-math id="Equ2_TeX">\documentclass[12pt]{minimal}
				\usepackage{amsmath}
				\usepackage{wasysym}
				\usepackage{amsfonts}
				\usepackage{amssymb}
				\usepackage{amsbsy}
				\usepackage{mathrsfs}
				\usepackage{upgreek}
				\setlength{\oddsidemargin}{-69pt}
				\begin{document} $$ \text{sim}\left(v_{1}, v_{2}\right)=\frac{v_{1} \cdot v_{2}}{\left\|v_{1}\right\|\left\|v_{2}\right\|}  $$ \end{document}</tex-math><graphic position="anchor" xlink:href="13326_2021_249_Article_Equ2.gif"/></alternatives></disp-formula></p><p>where <italic>v</italic><sub>1</sub> and <italic>v</italic><sub>2</sub> are two vectors representing two given entities.</p></sec></sec><sec id="Sec13" sec-type="results"><title>Results</title><sec id="Sec14"><title>Disease–phenotype datasets</title><p>We generated a total of four datasets covering disease–phenotype associations (see Fig. <xref rid="Fig1" ref-type="fig">1</xref>) by using text mining and semi-automatic collection of associations from public data resources. The “Text Mined” and “Semi-automated” datasets contain all of the text mined and semi-automatically gathered ICD-10-phenotype associations respectively. “Text Mined (UKB)” is the subset of “Text Mined” which covers the associations of only common diseases found in UK Biobank. On the other hand, “Semi-automated (UKB)” covers further manually curated known associations of these common diseases. Table <xref rid="Tab1" ref-type="table">1</xref> presents the distribution of the associations in the generated datasets based on their provenance. Our aim is specifically to associate common diseases in ICD-10 with phenotypes so that we can map datasets using ICD-10 to phenotypes.
<table-wrap id="Tab1"><label>Table 1</label><caption xml:lang="en"><p>Distribution of disease-phenotype associations in the generated datasets by provenance</p></caption><table frame="hsides" rules="groups"><thead><tr><th align="left"><p>Provenance</p></th><th align="left"><p>Text Mined</p></th><th align="left"><p>Text Mined (UKB)</p></th><th align="left"><p>Semi-automatic</p></th><th align="left"><p>Semi-automatic (UKB)</p></th></tr></thead><tbody><tr><td align="left"><p>PubMed</p></td><td align="left"><p>2,755,333</p></td><td align="left"><p>985,511</p></td><td align="left"><p>-</p></td><td align="left"><p>-</p></td></tr><tr><td align="left"><p>Wikidata</p></td><td align="left"><p>-</p></td><td align="left"><p>-</p></td><td align="left"><p>1,838</p></td><td align="left"><p>295</p></td></tr><tr><td align="left"><p>HPO (through OMIM–ICD-10 from Wikidata)</p></td><td align="left"><p>-</p></td><td align="left"><p>-</p></td><td align="left"><p>32,323</p></td><td align="left"><p>3,914</p></td></tr><tr><td align="left"><p>HPO (through OMIM–ICD-10 from UMLS)</p></td><td align="left"><p>-</p></td><td align="left"><p>-</p></td><td align="left"><p>2,362</p></td><td align="left"><p>423</p></td></tr><tr><td align="left"><p>UMLS</p></td><td align="left"><p>-</p></td><td align="left"><p>-</p></td><td align="left"><p>1,287</p></td><td align="left"><p>541</p></td></tr><tr><td align="left"><p>Expert curation</p></td><td align="left"><p>-</p></td><td align="left"><p>-</p></td><td align="left"><p>-</p></td><td align="left"><p>433</p></td></tr><tr><td align="left"><p>Propagation(ICD-10)</p></td><td align="left"><p>-</p></td><td align="left"><p>-</p></td><td align="left"><p>10,201</p></td><td align="left"><p>1,214</p></td></tr><tr><td align="left"><p>Propagation(HPO)</p></td><td align="left"><p>-</p></td><td align="left"><p>-</p></td><td align="left"><p>9660</p></td><td align="left"><p>756</p></td></tr><tr><td align="left"><p>TOTAL</p></td><td align="left"><p>2,755,333</p></td><td align="left"><p>985,511</p></td><td align="left"><p>57,671</p></td><td align="left"><p>7,576</p></td></tr></tbody></table><table-wrap-foot><p>UKB denotes the subset covering common diseases only from UK Biobank</p></table-wrap-foot></table-wrap></p><p>The Text Mined dataset covers a total of 2,755,333 positive disease–phenotype associations (NPMI score &gt; 0) between 13,610 distinct phenotype classes (from either MP or HP) and 6,263 distinct diseases (from ICD-10) from the literature. A total of 985,511 out of 2,755,333 disease–phenotype annotations can be linked to 1,557 of 2,106 common ICD-10 codes (Text Mined (UKB)). For the remaining 549 diseases, we could not find any positive association from the literature based on our approach.</p><p>The Semi-automatic dataset covers a total of 57,671 ICD-10–HPO associations among 7,610 distinct ICD-10 classes and 6,741 distinct phenotypes obtained by integrating a number of manually curated datasets (see Table <xref rid="Tab1" ref-type="table">1</xref>). Out of the 57,671 ICD-10–HPO associations, we gathered the majority of the associations (37,810 out of 57,671 associations, linked to 4,207 of the 7,610 ICD-10 classes) through resources covering rare or common diseases. We obtained a total of 1,838 association from Wikidata, 32,323 associations from the HPO database through OMIM–ICD-10 links from Wikidata and 2,362 through OMIM–ICD-10 links from UMLS; we also obtained 1,287 associations directly from UMLS. We gathered the remaining 19,861 associations (linked to 3,403 of the 7,610 ICD-10 classes) by propagating phenotype annotations of diseases from their subclasses in the ICD-10 hierarchy. We obtained 10,201 out of 19,861 associations by propagating phenotypes from their superclass based on the ICD-10 hierarchy; we obtained the remaining 9,660 out of 19,861 associations by lexical match between the superclass labels and the phenotype labels in HPO.</p><p>We sub-selected 2,106 distinct ICD-10 diseases from the Semi-automatic dataset covering all the common ICD-10 codes within UK Biobank. We curated their phenotype associations manually and filtered out the false positives. This curated dataset (Semi-automatic (UKB)) contains a total of 7,576 disease–phenotype associations gathered in a semi-automated way (see Materials and Methods) between 1,995 (of 2,106) common ICD-10 diseases and 2,757 distinct phenotypes linked to HPO. We gathered the majority of phenotype associations (4,337 out of 7,576 associations) for 334 distinct ICD-10 codes from HPO through ICD-10–OMIM links in either Wikidata (3,914/4,337 pairs) or UMLS (423/4,337 pairs). We gathered 541/7,576 associations linked to 473 distinct ICD-10 codes through direct mappings of ICD-10 and HPO in UMLS. We gathered 295/7,576 associations for 43 distinct ICD-10 codes from Wikidata. We generated 1,214/7,576 associations for 335 distinct ICD-10 codes by propagating phenotypes from their superclass based on the ICD-10 hierarchy.</p><p>We manually curated 433/7,576 disease–phenotype associations for 433 ICD-10 codes. We generated a total of 756/7576 associations linked to 483 ICD-10 codes by propagating phenotypes from their superclasses when we found a lexical match between the superclass labels and the phenotype labels in HPO.</p></sec><sec id="Sec15"><title>Phenotypic similarity of text mined and known associations</title><p>We measured the semantic similarity between our text mined and the known phenotypes of the diseases. There are 296 diseases in our dataset that are contained both in ICD-10 and OMIM and for which we can obtain phenotype associations both from our text mining approach and from curated data in the HPO database. We measured the semantic similarity between the phenotype profiles of a given disease by using cosine similarity between the ontology embeddings of the disease’s phenotype profiles generated through OWL2Vec* [<xref ref-type="bibr" rid="CR49">49</xref>].</p><p>Our Text Mined dataset consists of disease–phenotype associations and each association has a score that determines the association strength. Among the diseases in our dataset, between 1 and 2,592 phenotypes are positively associated. We assume that not all positive associations may be relevant but only the stronger associations provide useful information about a disease. We test this hypothesis by ranking phenotypes for each disease by their association (NPMI) score. We then include phenotypes in a disease–phenotype profile using varying thresholds for the number of phenotypes to include (based on the association score). To determine a threshold that yields a phenotype profile similar to manually curated ones, we compare the semantic similarity of the thresholded phenotype profiles to the manually curated profiles for the same disease; we evaluate the similarly using receiver operating characteristic (ROC) curves [<xref ref-type="bibr" rid="CR51">51</xref>]. We find that a threshold of 76 phenotypes results in maximal similarity to the manually curated disease–phenotype associations (ROCAUC 0.95). Figure <xref rid="Fig4" ref-type="fig">4</xref> shows the results of our experiment.
<fig id="Fig4"><label>Fig. 4</label><caption xml:lang="en"><p>AUC values obtained for the phenotypic similarity of text-mined and known diseases from HPO at different NPMI ranks</p></caption><p><graphic specific-use="HTML" mime-subtype="PNG" xlink:href="MediaObjects/13326_2021_249_Fig4_HTML.png" id="MO4"/></p></fig></p></sec><sec id="Sec16"><title>Predicting gene-disease associations</title><p>We further evaluated whether our Text Mined and Semi-automatic (UKB) datasets are useful in identifying gene–disease associations based on phenotype similarity. We found 53 diseases in ICD-10 that can be mapped directly to OMIM and are also present in our Text Mined and Semi-automatic (UKB) datasets. These 53 diseases are associated with 216 genes in our gene–disease dataset gathered from MGI.</p><p>Utilizing the text mined disease-phenotype associations with their association score, we followed a similar procedure as before and rank phenotypes for each disease based on their association score and vary the rank as threshold parameter. We then compared these phenotype profiles to phenotypes resulting from loss of function mouse models using the cosine similarity between their ontology embeddings, and evaluated how well this method recovers known gene–disease associations. Figure <xref rid="Fig5" ref-type="fig">5</xref> shows the resulting ROCAUC at different NPMI ranks. We find the maximal ROCAUC value at rank 74 (ROCAUC 0.62).
<fig id="Fig5"><label>Fig. 5</label><caption xml:lang="en"><p>AUC values obtained for the phenotypic similarity of text-mined diseases and known genes from MGI at different NPMI ranks</p></caption><p><graphic specific-use="HTML" mime-subtype="PNG" xlink:href="MediaObjects/13326_2021_249_Fig5_HTML.png" id="MO5"/></p></fig></p><p>We further used different datasets to find gene–disease associations through phenotype similarity: our Text Mined dataset with a threshold of 74 per disease; our Semi-automatic (UKB) dataset collected from multiple databases; the phenotypes associated with the 53 diseases in the HPO database; and combinations thereof. Figure <xref rid="Fig6" ref-type="fig">6</xref> shows the ROC curves resulting from this comparison. The ROCAUC values range from 0.79 for combining Text Mined and Semi-automatic (UKB) datasets to 0.62 for only the Text Mined dataset.
<fig id="Fig6"><label>Fig. 6</label><caption xml:lang="en"><p>Comparison of ROC curves for predicting gene–disease associations using cosine similarity</p></caption><p><graphic specific-use="HTML" mime-subtype="PNG" xlink:href="MediaObjects/13326_2021_249_Fig6_HTML.png" id="MO6"/></p></fig></p></sec><sec id="Sec17"><title>Comparison to expert-curated disease–phenotype associations</title><p>We created an expert-curated disease–phenotype association dataset to use for validation. This validation dataset consisted of 830 disease–phenotype associations for 53 diseases. To generate this dataset, we first gathered the semi-automatically curated ICD-10–HPO associations for these 53 diseases from our dataset. False positive HPO terms were filtered out and missing associations were added by an expert; 269 annotations were added. Because the HPO database contains mainly annotations to rare Mendelian diseases, most of the phenotype annotations contained in it are predicated on single gene, oligogenic, recurrent CNV or chromosome structural, disease etiology. While much of the phenotype annotation we need for common disease may be obtained from these annotations, the HPO data includes many phenotypes that are only found in the genetic syndromic disease and not in sporadic occurrences; this is discussed below. Consequently, in putting together the validation dataset, phenotypes which are not found in sporadic disease were treated as false positive unless the ICD class explicitly referred to an OMIM disease. In addition, high level terms such as HP:0002664<italic>Neoplasm</italic>, were excluded as being of low information content.</p><p>We used this corpus to evaluate the datasets we generated by comparing phenotype classes associated with diseases directly, using two types of evaluation, “strict” and “soft”. We called an evaluation strict if we ignored the hierarchy and semantics of phenotype ontologies and only compared whether phenotype classes matched exactly between our dataset and our benchmark. In the soft evaluation, we first propagated disease–phenotype associations over the phenotype ontology hierarchy and then evaluated on all levels of the ontology.</p><p>Our semi-automatically curated dataset covered a total of 649 disease–phenotype associations for those 53 diseases. 568/649 of the associations were true positives, 81/649 were false positives. We missed a total of 262/830 annotations (false negatives). We estimated the Precision as 0.88, Recall as 0.68 and F-score as 0.77.</p><p>Figure <xref rid="Fig7" ref-type="fig">7</xref> shows the performance analysis of the text mining extracts against the validation dataset. The performance of the text mining process varied over different NPMI ranks. Max F-score value of 0.21 was achieved at NPMI rank 16.
<fig id="Fig7"><label>Fig. 7</label><caption xml:lang="en"><p>Performance analysis of text mining against the validation dataset over different NPMI ranks (strict)</p></caption><p><graphic specific-use="HTML" mime-subtype="PNG" xlink:href="MediaObjects/13326_2021_249_Fig7_HTML.png" id="MO7"/></p></fig></p><p>We have a total of 3,499 disease-phenotype annotations in the validation dataset when we propagate annotations based on the PhenomeNET ontology. On the other hand, our semi-automatically curated dataset covers a total of 2,830 disease-phenotype annotations after the propagation process. In the “soft” settings, we found that 2,454/2,830 associations are true positive, 376/2,830 are false positive, and 1,045/3,499 are false negative. We estimated the Precision as 0.87, Recall as 0.70 and F-score as 0.78.</p><p>Figure <xref rid="Fig8" ref-type="fig">8</xref> shows the performance analysis of the text mined extracts against the validation dataset under the “soft” settings. The performance of the text mining process varies over different NPMI ranks. The best F-score is achieved at the NPMI rank of 27 as a value of 0.44.
<fig id="Fig8"><label>Fig. 8</label><caption xml:lang="en"><p>Performance analysis of text mining against the validation dataset over different NPMI ranks (soft)</p></caption><p><graphic specific-use="HTML" mime-subtype="PNG" xlink:href="MediaObjects/13326_2021_249_Fig8_HTML.png" id="MO8"/></p></fig></p></sec><sec id="Sec18"><title>Coverage of the generated datasets</title><p>There are a total of 19,133 distinct ICD-10 codes. We linked 6,263 and 7,610 ICD-10 codes to their phenotypes by using text mining and the semi-automatic strategy, respectively. While we linked 4,118 ICD-10 classes to their phenotypes by both of the methods (overlap); 9,755 (51%) ICD-10 classes were linked to their phenotypes by either methods. Hence, we were unable to link 9,378 (49%) ICD-10 classes to their phenotypes. We discuss the main reasons of being unable to link these ICD-10 classes to their phenotypes in detail in the Discussion section.</p></sec><sec id="Sec19"><title>Error analysis</title><sec id="Sec20"><title>Semi-automatically curated data</title><p>We identified a total of 1,369 false positives during the semi-automatic curation of the associations from all of the 2,106 common diseases. We found that, while 963/1,369 false positives were due to the associations from existing resources, the remaining 406/1,369 false positives were due to the propagation of the annotations. 170/406 false positives are due to their lexical superclass matches in the HPO dataset and 236/406 false positives are due to their ICD-10 superclass-based annotation propagation. For example, ICD-10:C43.5<italic>Malignant melanoma of the trunk</italic> produced the annotation to HP:0007716<italic>Uveal melanoma</italic>, due to propagation from ICD-10:C43, <italic>Malignant melanoma of skin</italic>. We gathered the association between ICD-10:C43 and HP:0007716 from the HPO database through the mapping between OMIM:155600–ICD-10:C43 from UMLS.</p><p>Further breaking down the 963 false positives generated from the known data, we found that 12/963 false positives were from the Wikidata set, 3/963 false positives were due to the ICD-10–HPO direct mappings in UMLS, 19/963 false positives were due to incorrect associations found during the manual expert curation due to inclusion of syndromic phenotypes as discussed above, and the remaining 929/963 false positives were due to the use of the asserted disease–phenotype annotations in the HPO database. We further investigated these 929 false positives. As the diseases and phenotypes are mapped to their OMIM and HPO identifiers, respectively, to obtain ICD-10 identifiers for the OMIM diseases, we investigated the portions of the false positives introduced through OMIM–ICD-10 mappings in UMLS and Wikidata. We found that 44/929 false positives were introduced due to OMIM–ICD-10 mappings from UMLS and the remaining 885/929 false positives, which constitute the majority, were introduced due to the OMIM–ICD-10 mappings from Wikidata.</p><p>For example ICD-10:I77.1, <italic>Stricture of artery</italic>, is annotated to HP:0002036, <italic>Hiatus hernia</italic>, because Wikidata maps this ICD-10 class to OMIM:208050, <italic>Arterial tortuosity syndrome</italic>, which has a wide clinical phenotype spectrum among which is <italic>Hiatal hernia</italic>. Phenotypes that would not normally be considered a manifestation of sporadic non-syndromic arterial stricture, such as <italic>Arachnodactyly</italic> or <italic>Hiatus hernia</italic> were considered false positives. However, correct annotations to HPO were obtained directly from UMLS, which provides a correct annotation HP:0100545, <italic>Arterial stenosis</italic>. In general, ICD-10 to OMIM mappings through Wikidata-generated candidate HPO annotations are associated with Mendelian, syndromic disease, accounting for the high number of false positives through this route. These had to be manually removed on a case-by-case basis using expert judgement, where sporadic disease would not be expected to have these associations.</p><p>False negatives, i.e. missing annotations, were called usually when the annotation was sparse but there are clear associated phenotypes available in HPO. The causes of this are interesting. For example HP:0000979, <italic>Purpura</italic>, was missing from the annotation to ICD-10:M31.3<italic>Wegener granulomatosis</italic> [<xref ref-type="bibr" rid="CR52">52</xref>] and HP:0025188, <italic>Retinal vasculitis</italic> missing from systemic ICD-10:M32.9<italic>Lupus erythematosus</italic> [<xref ref-type="bibr" rid="CR53">53</xref>]. In the former case, although <italic>Wegener granulomatosis</italic> is in OMIM (OMIM:608710), there is no clinical synopsis and it was therefore not possible to gather annotations from the HPO database. For the latter, <italic>Systemic lupus erythematosus</italic>, HP:0002725 is treated as a “bundled term” phenotype in the HPO database and therefore no more granular phenotype annotations are available. There are no direct HPO annotations for <italic>Systemic lupus erythematosis</italic> in UMLS. We cannot provide any assurance that all of the possible missing annotations have been added to the dataset, but have provided best efforts with the resources available. We hope that users might over time request the addition of phenotypes to their diseases of interest.</p></sec><sec id="Sec21"><title>Text mined data</title><p>For the analysis of the text mined associations, we used the extracts generated based on the NPMI rank 16 which gave us the best result on the validation dataset by using the strict evaluation (precision 0.25, recall 0.17, and F-score 0.21). We have a total of 568 ICD-10–HPO pairs in this text-mined dataset. We found that 143/568 are true positives and 425/568 are false positives. We missed a total of 687 associations (false negatives). Our manual analysis on the 425 false positives show that only a small portion of them (47/425) are false positives and the majority of them (376/425) are actually true positive associations which are not covered by our validation dataset. Our validation dataset includes only the obvious and distinguishing phenotypes of diseases. These 376 associations are the associations of the diseases with the high level of HPO classes. For example, <italic>Malignant neoplasm of stomach, unspecified</italic> (ICD-10:C16.9) is associated with <italic>Neoplasm</italic> (HP:0002664) according to our text mining extracts. This is a true positive by manual analysis but was counted as a FP since it is not covered within our validation dataset as <italic>Neoplasm</italic> is a high level phenotype for all malignant and benign proliferative lesions and of low information content. The false positives are mainly due to the co-mentions of associated disease concepts, or negations in the publications (X is not a Y). Some examples of such associations include <italic>Acute myeloid leukaemia</italic> (ICD-10:C92.0) and Chronic myelomonocytic leukemia (HP:0012325) as well as Primary open-angle glaucoma (ICD-10:H40.1) and <italic>Angle closure glaucoma</italic> (HP:0012109). Analysis of the 687 false negative samples showed that actually 473 of 687 pairs (69%) have been extracted from the literature but they do not rank in the top 16 based on their NPMI scores of association strength. The other missing ones are mainly due to weak or no evidence in the literature. For example, there are no publications mentioning <italic>Marfan syndrome</italic> (ICD-10:Q87.4) and <italic>Decreased muscle mass</italic> (HP:0003199); there are only 2 publications mentioning Parkinson’s disease (ICD-10:G20) and <italic>Macrocephaly</italic> (HP:0000256) in title or abstract together in PubMed (search was done on 15th April 2021). One of the publications is published in 2021 which is not covered by our current dataset. Therefore, there is no significant supporting evidence in the literature to infer a positive association between the given disease–phenotype pairs. Other false negatives could be due to the missing disease/phenotype synonyms. Altogether, we estimated the actual performance of the text mining method (at the NPMI rank 16) as an F-score value of 0.59, a precision of 0.92 and a recall of 0.43.</p></sec></sec></sec><sec id="Sec22" sec-type="discussion"><title>Discussion</title><p>We have previously reported an extensive annotation of the diseases in DO based on a text mining analysis of PubMed abstracts and titles [<xref ref-type="bibr" rid="CR30">30</xref>]. This included phenotype annotations to 6,000 common, rare and infectious diseases of which 4,768 are diseases from OMIM [<xref ref-type="bibr" rid="CR29">29</xref>]. The under-representation of sporadic, common or complex disease in this dataset and the fact that DO is not frequently used in routine clinical recording were the motivation to develop a set of HPO annotations to terms in the much larger ICD-10 terminology. Here we have carried out a large-scale text mining analysis of PubMed using term labels, synonyms and acronyms of ICD-10 codes, and augmented this new analysis with data from three publicly available annotation sources, UMLS, Wikidata and the HPO database.</p><p>While Wikidata and the HPO database contain almost exclusively phenotypes for rare diseases found in OMIM and Orphanet, they present a source of annotation that may be exploited for common disease as explained below. A similar but more limited approach to phenotypic annotation for common disease was implemented by Sarntivijai et al. [<xref ref-type="bibr" rid="CR54">54</xref>] using ontology-driven literature mining for two classes of disease, <italic>Inflammatory bowel disease</italic> and Autoimmune disease, together with their subclasses in the Experimental Factor Ontology (EFO) [<xref ref-type="bibr" rid="CR55">55</xref>]. This produced 1,452 and 2,810 disease–phenotype pairs for inflammatory bowel disease (IBD) and autoimmune disease of which 41.6% candidate IBD phenotype associations were deemed correct by manual review. Similar to the strategy we take here, the authors of the study removed non-informative phenotypes such as “All”, “Chronic”, or “Death” but unlike us excluded classes in HPO that were deemed to represent disease entities, using expert judgement. The authors discuss some of the problems we also encountered of annotation validation on existing datasets.</p><p>In attempting a large scale phenotypic annotation of a significant number of the disease concepts in ICD-10, we have noted several issues. In trying to semi-automatically generate this corpus of annotations, one question is the decision as to what should be considered as part of a phenotypic manifestations, what level of granularity should be used, and the reliability of existing sources of annotation such as Wikidata, the HPO database, and UMLS. The definition of a phenotype as an observable characteristic covers simple signs and symptoms, and syndromic manifestations, but operationally “phenotypes” are included in the HPO database that may occur in isolation as “diseases” such as <italic>Diabetes</italic> or Tetralogy of Fallot (HPO regards these as “bundled phenotypes” and are included for pragmatic reasons). The decision as to how to select our annotation strategy can therefore only be guided by the purposes for which these annotations are developed, and by the best outcome on evaluation. We believe that the inclusive approach we take provides a valid strategy as assessed by performance on disease/gene prediction from the MGI dataset.</p><p>We find that many rare and rather few common diseases are extensively and accurately annotated. In some cases this is due to the deep annotation in the OMIM/HPO databases, UMLS, and, to a lesser extent, in Wikidata. The mapping of ICD classes to HPO involves for the most part working through the intermediary mappings to OMIM given in UMLS or Wikidata. As discussed above, this often results in phenotype annotations designed to describe rare inherited diseases or syndromes and not common or sporadic diseases. Although ICD classes sometimes include rare diseases explicitly, most do not, and therefore the intention in annotating a patient to an ICD-10 class is that of noting common/sporadic disease unless rare disease is asserted in the ICD-10 class chosen. As a consequence, we expertly edited annotations from HPO to align with the sporadic/common disease implied by the ICD class, giving rise to an increased number of false positive calls. We did not edit when the ICD class explicitly included an OMIM disease. This process, while driven by expert opinion is nevertheless subjective and represents a potential weakness in our approach. The low recall versus high specificity we obtain in recovering MGI gene disease associations is a consequence of disease annotation in MGI being to OMIM diseases when we edited OMIM disease phenotype annotations to approach the less complex annotation expected of sporadic disease. Our validation approach is therefore limited by what annotation datasets are available, and in the absence of any other manually curated large disease/phenotype datasets we believe that this is the best approach currently available, while not optimal.</p><p>We attempted to evaluate how removing some of these deeply annotated diseases affected the validation and found overall small changes in evaluation performance. More specifically, we identified that there are 3 heavily annotated diseases out of 53 diseases in the validation dataset <italic>Marfan’s disease</italic> (ICD-10:Q87.4), hereditary hemorrhagic telangiectasis (ICD-10:I78.0), and <italic>hereditary factor VIII deficiency</italic> (ICD-10:D66). When we removed these 3 diseases from the evaluation, the performance of the semi-automatic curation drops from an F-score value of 0.77 to 0.73.</p><p>Regarding the coverage of the datasets generated, we were unable to link 49% (9,378 out of 19,133) ICD-10 classes to their phenotypes either by semi-automatic or text mining methods. The majority of the missing ICD-10 terms are <italic>Diseases of the musculoskeletal system and connective tissue</italic>; ICD-10:M00–M99 (2574 ICD-10 codes), Injury, poisoning and certain other consequences of external causes, ICD-10:S00–T88 (1297 ICD-10 codes) and <italic>External causes of morbidity</italic>; ICD-10:V00–Y99 covering ICD-10:X00-99 (1113 ICD-10 codes), ICD-10:W00–W99 (1060 ICD-10 codes), ICD-10:V00–V99 (909 ICD-10 codes) and ICD10:Y00–Y99 (635 ICD-10 codes). We miss linking these ICD-10 codes to their phenotypes due to several methodological issues as well as the data available in the resources (HPO, UMLS, Wikidata, PubMed). More specifically, we text mined ICD-10–phenotype associations from the PubMed abstracts only and full-text articles are not covered in this study, which potentially include more associations. Furthermore, we miss some association of diseases which have long labels (e.g. ICD-10:Z62.6, <italic>Inappropriate parental pressure and other abnormal qualities of upbringing</italic>; ICD-10:X44, <italic>Accidental poisoning by and exposure to other and unspecified drugs, medicaments and biological substances</italic>) and therefore they are very unlikely to be mentioned in titles or abstracts in full. In addition, some of the associations are missed due to their low NPMI signal based on our method (we considered associations having NPMI &gt; 0). These missing ICD-10 codes cover mainly injuries, poisoning and infectious diseases which are not focus of HPO and the other resources used in this study. Therefore, lack of these classes is not likely to reduce the utility of the generated datasets for the purposes motivating their development, which is to link phenotypes to genetic variants and underlying molecular processes.</p><p>A well established problem is that for an instance of a disease in an individual patient all phenotypes will not necessarily be present and will evolve with time. A weakness of our annotation model is that phenotype associations are treated as a “bag of phenotypes” which lacks precision and flexibility. Future work will look at application of an Ontology of Biomedical AssociatioN (OBAN) data model to our results, which allows for the inclusion of qualification into the association between disease and phenotype [<xref ref-type="bibr" rid="CR54">54</xref>].</p></sec><sec id="Sec23" sec-type="conclusion"><title>Conclusion</title><p>We used a semi-automatic and a text mining based method to create four datasets of disease–phenotype associations. The generated disease–phenotype associations are useful for completing the phenotype profiles of the diseases linked to clinical resources, and can be used to investigate gene–disease associations. All the data is publicly available at Zenodo (DOI:<ext-link xlink:href="https://doi.org/10.5281/zenodo.4726714" ext-link-type="doi">https://doi.org/10.5281/zenodo.4726714</ext-link>) for community use.</p></sec></body><back><ack><title>Acknowledgements</title><p>This research has been conducted using the UK Biobank Resource under the Application Number 31224.</p></ack><sec sec-type="author-contribution"><title>Authors’ contributions</title><p>PS and RH conceived the experiments; ŞK conducted all the experiments except phenotypic similarity measurements; SA conducted phenotype similarity experiments; all manual annotations are done by PS; RH, PS, and ŞK analysed the results, ŞK drafted the initial version of the manuscript, SA, PS, RH, GVG revised the manuscript. PS, GVG, RH acquired funding to support this work. All authors reviewed and approved the final version of the manuscript.</p></sec><sec><title>Funding</title><p>This study is supported by King Abdullah University of Science and Technology (KAUST) Office of Sponsored Research (OSR) under Award No. URF/1/3790-01-01, URF/1/4355-01-01, FCC/1/1976-28-01, and FCC/1/1976-29-01. PNS acknowledges the support of The Alan Turing Institute.</p></sec><sec sec-type="data-availability"><title>Availability of data and materials</title><p>We make all data freely available at <ext-link xlink:href="https://doi.org/10.5281/zenodo.4726713" ext-link-type="doi">https://doi.org/10.5281/zenodo.4726713</ext-link>.</p><p>We make the source code developed available from Github, <ext-link xlink:href="https://github.com/bio-ontology-research-group/icdpheno" ext-link-type="url">https://github.com/bio-ontology-research-group/icdpheno</ext-link></p></sec><sec sec-type="ethics-statement"><sec id="FPar1"><title>Ethics approval and consent to participate</title><p>Not applicable.</p></sec><sec id="FPar2"><title>Consent for publication</title><p>Not applicable.</p></sec><sec id="FPar3" sec-type="COI-statement"><title>Competing interests</title><p>The authors declare that they have no competing interests.</p></sec></sec><ref-list id="Bib1"><title>References</title><ref-list><ref id="CR1"><label>1</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Collins</surname><given-names>FS</given-names></name><name><surname>Doudna</surname><given-names>JA</given-names></name><name><surname>Lander</surname><given-names>ES</given-names></name><name><surname>Rotimi</surname><given-names>CN</given-names></name></person-group><article-title xml:lang="en">Human molecular genetics and genomics — important advances and exciting possibilities</article-title><source>N Engl J Med</source><year>2021</year><volume>384</volume><issue>1</issue><fpage>1</fpage><lpage>4</lpage><pub-id pub-id-type="doi">10.1056/NEJMp2030694</pub-id><comment>https://doi.org/10.1056/nejmp2030694</comment></mixed-citation></ref><ref id="CR2"><label>2</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fernandez-Marmiesse</surname><given-names>A</given-names></name><name><surname>Gouveia</surname><given-names>S</given-names></name><name><surname>Couce</surname><given-names>ML</given-names></name></person-group><article-title xml:lang="en">NGS technologies as a turning point in rare disease research, diagnosis and treatment</article-title><source>Curr Med Chem</source><year>2018</year><volume>25</volume><issue>3</issue><fpage>404</fpage><lpage>32</lpage><pub-id pub-id-type="doi">10.2174/0929867324666170718101946</pub-id><comment>https://doi.org/10.2174/0929867324666170718101946</comment></mixed-citation></ref><ref id="CR3"><label>3</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Eichler</surname><given-names>EE</given-names></name></person-group><article-title xml:lang="en">Genetic variation, comparative genomics, and the diagnosis of disease</article-title><source>N Engl J Med</source><year>2019</year><volume>381</volume><issue>1</issue><fpage>64</fpage><lpage>74</lpage><pub-id pub-id-type="doi">10.1056/NEJMra1809315</pub-id><comment>https://doi.org/10.1056/nejmra1809315</comment></mixed-citation></ref><ref id="CR4"><label>4</label><mixed-citation publication-type="other">Rehm HL, Fowler DM. Keeping up with the genomes: scaling genomic variant interpretation. Genome Med. 2019; 12(1). <ext-link xlink:href="https://doi.org/10.1186/s13073-019-0700-4" ext-link-type="doi">https://doi.org/10.1186/s13073-019-0700-4</ext-link>.</mixed-citation></ref><ref id="CR5"><label>5</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Crouch</surname><given-names>DJM</given-names></name><name><surname>Bodmer</surname><given-names>WF</given-names></name></person-group><article-title xml:lang="en">Polygenic inheritance, GWAS, polygenic risk scores, and the search for functional variants</article-title><source>Proc Natl Acad Sci</source><year>2020</year><volume>117</volume><issue>32</issue><fpage>18924</fpage><lpage>33</lpage><pub-id pub-id-type="doi">10.1073/pnas.2005634117</pub-id><comment>https://doi.org/10.1073/pnas.2005634117</comment></mixed-citation></ref><ref id="CR6"><label>6</label><mixed-citation publication-type="other">Hartin SN, Means JC, Alaimo JT, Younger ST. Expediting rare disease diagnosis: a call to bridge the gap between clinical and functional genomics. Mol Med. 2020; 26(1). <ext-link xlink:href="https://doi.org/10.1186/s10020-020-00244-5" ext-link-type="doi">https://doi.org/10.1186/s10020-020-00244-5</ext-link>.</mixed-citation></ref><ref id="CR7"><label>7</label><mixed-citation publication-type="other">Cano-Gamez E, Trynka G. From GWAS to function: Using functional genomics to identify the mechanisms underlying complex diseases. Front Genet. 2020; 11. <ext-link xlink:href="https://doi.org/10.3389/fgene.2020.00424" ext-link-type="doi">https://doi.org/10.3389/fgene.2020.00424</ext-link>.</mixed-citation></ref><ref id="CR8"><label>8</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hartley</surname><given-names>T</given-names></name><name><surname>Lemire</surname><given-names>G</given-names></name><name><surname>Kernohan</surname><given-names>KD</given-names></name><name><surname>Howley</surname><given-names>HE</given-names></name><name><surname>Adams</surname><given-names>DR</given-names></name><name><surname>Boycott</surname><given-names>KM</given-names></name></person-group><article-title xml:lang="en">New diagnostic approaches for undiagnosed rare genetic diseases</article-title><source>Annu Rev Genomics Hum Genet</source><year>2020</year><volume>21</volume><issue>1</issue><fpage>351</fpage><lpage>72</lpage><pub-id pub-id-type="doi">10.1146/annurev-genom-083118-015345</pub-id><comment>https://doi.org/10.1146/annurev-genom-083118-015345</comment></mixed-citation></ref><ref id="CR9"><label>9</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tam</surname><given-names>V</given-names></name><name><surname>Patel</surname><given-names>N</given-names></name><name><surname>Turcotte</surname><given-names>M</given-names></name><name><surname>Bossé</surname><given-names>Y</given-names></name><name><surname>Paré</surname><given-names>G</given-names></name><name><surname>Meyre</surname><given-names>D</given-names></name></person-group><article-title xml:lang="en">Benefits and limitations of genome-wide association studies</article-title><source>Nat Rev Genet</source><year>2019</year><volume>20</volume><issue>8</issue><fpage>467</fpage><lpage>84</lpage><pub-id pub-id-type="doi">10.1038/s41576-019-0127-1</pub-id><comment>https://doi.org/10.1038/s41576-019-0127-1</comment></mixed-citation></ref><ref id="CR10"><label>10</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Köhler</surname><given-names>S</given-names></name><name><surname>Gargano</surname><given-names>M</given-names></name><name><surname>Matentzoglu</surname><given-names>N</given-names></name><name><surname>Carmody</surname><given-names>LC</given-names></name><name><surname>Lewis-Smith</surname><given-names>D</given-names></name><name><surname>Vasilevsky</surname><given-names>NA</given-names></name><name><surname>Danis</surname><given-names>D</given-names></name><name><surname>Balagura</surname><given-names>G</given-names></name><name><surname>Baynam</surname><given-names>G</given-names></name><name><surname>Brower</surname><given-names>AM</given-names></name><name><surname>Callahan</surname><given-names>TJ</given-names></name><name><surname>Chute</surname><given-names>CG</given-names></name><name><surname>Est</surname><given-names>JL</given-names></name><name><surname>Galer</surname><given-names>PD</given-names></name><name><surname>Ganesan</surname><given-names>S</given-names></name><name><surname>Griese</surname><given-names>M</given-names></name><name><surname>Haimel</surname><given-names>M</given-names></name><name><surname>Pazmandi</surname><given-names>J</given-names></name><name><surname>Hanauer</surname><given-names>M</given-names></name><name><surname>Harris</surname><given-names>NL</given-names></name><name><surname>Hartnett</surname><given-names>MJ</given-names></name><name><surname>Hastreiter</surname><given-names>M</given-names></name><name><surname>Hauck</surname><given-names>F</given-names></name><name><surname>He</surname><given-names>Y</given-names></name><name><surname>Jeske</surname><given-names>T</given-names></name><name><surname>Kearney</surname><given-names>H</given-names></name><name><surname>Kindle</surname><given-names>G</given-names></name><name><surname>Klein</surname><given-names>C</given-names></name><name><surname>Knoflach</surname><given-names>K</given-names></name><name><surname>Krause</surname><given-names>R</given-names></name><name><surname>Lagorce</surname><given-names>D</given-names></name><name><surname>McMurry</surname><given-names>JA</given-names></name><name><surname>Miller</surname><given-names>JA</given-names></name><name><surname>Munoz-Torres</surname><given-names>MC</given-names></name><name><surname>Peters</surname><given-names>RL</given-names></name><name><surname>Rapp</surname><given-names>CK</given-names></name><name><surname>Rath</surname><given-names>AM</given-names></name><name><surname>Rind</surname><given-names>SA</given-names></name><name><surname>Rosenberg</surname><given-names>AZ</given-names></name><name><surname>Segal</surname><given-names>MM</given-names></name><name><surname>Seidel</surname><given-names>MG</given-names></name><name><surname>Smedley</surname><given-names>D</given-names></name><name><surname>Talmy</surname><given-names>T</given-names></name><name><surname>Thomas</surname><given-names>Y</given-names></name><name><surname>Wiafe</surname><given-names>SA</given-names></name><name><surname>Xian</surname><given-names>J</given-names></name><name><surname>Yüksel</surname><given-names>Z</given-names></name><name><surname>Helbig</surname><given-names>I</given-names></name><name><surname>Mungall</surname><given-names>CJ</given-names></name><name><surname>Haendel</surname><given-names>MA</given-names></name><name><surname>Robinson</surname><given-names>PN</given-names></name></person-group><article-title xml:lang="en">The human phenotype ontology in 2021</article-title><source>Nucleic Acids Res</source><year>2020</year><volume>49</volume><issue>D1</issue><fpage>1207</fpage><lpage>17</lpage><pub-id pub-id-type="doi">10.1093/nar/gkaa1043</pub-id><comment>https://doi.org/10.1093/nar/gkaa1043</comment></mixed-citation></ref><ref id="CR11"><label>11</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gkoutos</surname><given-names>GV</given-names></name><name><surname>Schofield</surname><given-names>PN</given-names></name><name><surname>Hoehndorf</surname><given-names>R</given-names></name></person-group><article-title xml:lang="en">The anatomy of phenotype ontologies: principles, properties and applications</article-title><source>Brief Bioinform</source><year>2017</year><volume>19</volume><issue>5</issue><fpage>1008</fpage><lpage>21</lpage><pub-id pub-id-type="doi">10.1093/bib/bbx035</pub-id><comment>https://doi.org/10.1093/bib/bbx035</comment></mixed-citation></ref><ref id="CR12"><label>12</label><mixed-citation publication-type="other">Rodríguez-García MÁ, Gkoutos GV, Schofield PN, Hoehndorf R. Integrating phenotype ontologies with PhenomeNET. J Biomed Semant. 2017; 8(1). <ext-link xlink:href="https://doi.org/10.1186/s13326-017-0167-4" ext-link-type="doi">https://doi.org/10.1186/s13326-017-0167-4</ext-link>.</mixed-citation></ref><ref id="CR13"><label>13</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Shefchek</surname><given-names>KA</given-names></name><name><surname>Harris</surname><given-names>NL</given-names></name><name><surname>Gargano</surname><given-names>M</given-names></name><name><surname>Matentzoglu</surname><given-names>N</given-names></name><name><surname>Unni</surname><given-names>D</given-names></name><name><surname>Brush</surname><given-names>M</given-names></name><name><surname>Keith</surname><given-names>D</given-names></name><name><surname>Conlin</surname><given-names>T</given-names></name><name><surname>Vasilevsky</surname><given-names>N</given-names></name><name><surname>Zhang</surname><given-names>XA</given-names></name><name><surname>Balhoff</surname><given-names>JP</given-names></name><name><surname>Babb</surname><given-names>L</given-names></name><name><surname>Bello</surname><given-names>SM</given-names></name><name><surname>Blau</surname><given-names>H</given-names></name><name><surname>Bradford</surname><given-names>Y</given-names></name><name><surname>Carbon</surname><given-names>S</given-names></name><name><surname>Carmody</surname><given-names>L</given-names></name><name><surname>Chan</surname><given-names>LE</given-names></name><name><surname>Cipriani</surname><given-names>V</given-names></name><name><surname>Cuzick</surname><given-names>A</given-names></name><name><surname>Della Rocca</surname><given-names>M</given-names></name><name><surname>Dunn</surname><given-names>N</given-names></name><name><surname>Essaid</surname><given-names>S</given-names></name><name><surname>Fey</surname><given-names>P</given-names></name><name><surname>Grove</surname><given-names>C</given-names></name><name><surname>Gourdine</surname><given-names>J-P</given-names></name><name><surname>Hamosh</surname><given-names>A</given-names></name><name><surname>Harris</surname><given-names>M</given-names></name><name><surname>Helbig</surname><given-names>I</given-names></name><name><surname>Hoatlin</surname><given-names>M</given-names></name><name><surname>Joachimiak</surname><given-names>M</given-names></name><name><surname>Jupp</surname><given-names>S</given-names></name><name><surname>Lett</surname><given-names>KB</given-names></name><name><surname>Lewis</surname><given-names>SE</given-names></name><name><surname>McNamara</surname><given-names>C</given-names></name><name><surname>Pendlington</surname><given-names>ZM</given-names></name><name><surname>Pilgrim</surname><given-names>C</given-names></name><name><surname>Putman</surname><given-names>T</given-names></name><name><surname>Ravanmehr</surname><given-names>V</given-names></name><name><surname>Reese</surname><given-names>J</given-names></name><name><surname>Riggs</surname><given-names>E</given-names></name><name><surname>Robb</surname><given-names>S</given-names></name><name><surname>Roncaglia</surname><given-names>P</given-names></name><name><surname>Seager</surname><given-names>J</given-names></name><name><surname>Segerdell</surname><given-names>E</given-names></name><name><surname>Similuk</surname><given-names>M</given-names></name><name><surname>Storm</surname><given-names>AL</given-names></name><name><surname>Thaxon</surname><given-names>C</given-names></name><name><surname>Thessen</surname><given-names>A</given-names></name><name><surname>Jacobsen</surname><given-names>JOB</given-names></name><name><surname>McMurry</surname><given-names>JA</given-names></name><name><surname>Groza</surname><given-names>T</given-names></name><name><surname>Köhler</surname><given-names>S</given-names></name><name><surname>Smedley</surname><given-names>D</given-names></name><name><surname>Robinson</surname><given-names>PN</given-names></name><name><surname>Mungall</surname><given-names>CJ</given-names></name><name><surname>Haendel</surname><given-names>MA</given-names></name><name><surname>Munoz-Torres</surname><given-names>MC</given-names></name><name><surname>Osumi-Sutherland</surname><given-names>D</given-names></name></person-group><article-title xml:lang="en">The Monarch Initiative in 2019: an integrative data and analytic platform connecting phenotypes to genotypes across species</article-title><source>Nucleic Acids Res</source><year>2019</year><volume>48</volume><issue>D1</issue><fpage>704</fpage><lpage>15</lpage><pub-id pub-id-type="doi">10.1093/nar/gkz997</pub-id><comment>http://dx.doi.org/10.1093/nar/gkz997. https://academic.oup.com/nar/article-pdf/48/D1/D704/32788250/gkz997.pdf</comment></mixed-citation></ref><ref id="CR14"><label>14</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Boudellioua</surname><given-names>I</given-names></name><name><surname>Kulmanov</surname><given-names>M</given-names></name><name><surname>Schofield</surname><given-names>PN</given-names></name><name><surname>Gkoutos</surname><given-names>GV</given-names></name><name><surname>Hoehndorf</surname><given-names>R</given-names></name></person-group><article-title xml:lang="en">Oligopvp: Phenotype-driven analysis of individual genomic information to prioritize oligogenic disease variants</article-title><source>Sci Rep</source><year>2018</year><volume>8</volume><issue>1</issue><fpage>14681</fpage><pub-id pub-id-type="doi">10.1038/s41598-018-32876-3</pub-id><comment>https://doi.org/10.1038/s41598-018-32876-3</comment></mixed-citation></ref><ref id="CR15"><label>15</label><mixed-citation publication-type="other">Cipriani V, Pontikos N, Arno G, Sergouniotis PI, Lenassi E, Thawong P, Danis D, Michaelides M, Webster AR, Moore AT, Robinson PN, Jacobsen JOB, Smedley D. An improved phenotype-driven tool for rare mendelian variant prioritization: Benchmarking exomiser on real patient whole-exome data. Genes. 2020; 11(4). <ext-link xlink:href="https://doi.org/10.3390/genes11040460" ext-link-type="doi">https://doi.org/10.3390/genes11040460</ext-link>.</mixed-citation></ref><ref id="CR16"><label>16</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zemojtel</surname><given-names>T</given-names></name><name><surname>Köhler</surname><given-names>S</given-names></name><name><surname>Mackenroth</surname><given-names>L</given-names></name><name><surname>Jäger</surname><given-names>M</given-names></name><name><surname>Hecht</surname><given-names>J</given-names></name><name><surname>Krawitz</surname><given-names>P</given-names></name><name><surname>Graul-Neumann</surname><given-names>L</given-names></name><name><surname>Doelken</surname><given-names>S</given-names></name><name><surname>Ehmke</surname><given-names>N</given-names></name><name><surname>Spielmann</surname><given-names>M</given-names></name><name><surname>Øien</surname><given-names>NC</given-names></name><name><surname>Schweiger</surname><given-names>MR</given-names></name><name><surname>Krüger</surname><given-names>U</given-names></name><name><surname>Frommer</surname><given-names>G</given-names></name><name><surname>Fischer</surname><given-names>B</given-names></name><name><surname>Kornak</surname><given-names>U</given-names></name><name><surname>Flöttmann</surname><given-names>R</given-names></name><name><surname>Ardeshirdavani</surname><given-names>A</given-names></name><name><surname>Moreau</surname><given-names>Y</given-names></name><name><surname>Lewis</surname><given-names>SE</given-names></name><name><surname>Haendel</surname><given-names>M</given-names></name><name><surname>Smedley</surname><given-names>D</given-names></name><name><surname>Horn</surname><given-names>D</given-names></name><name><surname>Mundlos</surname><given-names>S</given-names></name><name><surname>Robinson</surname><given-names>PN</given-names></name></person-group><article-title xml:lang="en">Effective diagnosis of genetic disease by computational phenotype analysis of the disease-associated genome</article-title><source>Sci Transl Med</source><year>2014</year><volume>6</volume><issue>252</issue><fpage>252</fpage><lpage>123</lpage><pub-id pub-id-type="doi">10.1126/scitranslmed.3009262</pub-id><comment>https://doi.org/10.1126/scitranslmed.3009262</comment></mixed-citation></ref><ref id="CR17"><label>17</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wakap</surname><given-names>SN</given-names></name><name><surname>Lambert</surname><given-names>DM</given-names></name><name><surname>Olry</surname><given-names>A</given-names></name><name><surname>Rodwell</surname><given-names>C</given-names></name><name><surname>Gueydan</surname><given-names>C</given-names></name><name><surname>Lanneau</surname><given-names>V</given-names></name><name><surname>Murphy</surname><given-names>D</given-names></name><name><surname>Cam</surname><given-names>YL</given-names></name><name><surname>Rath</surname><given-names>A</given-names></name></person-group><article-title xml:lang="en">Estimating cumulative point prevalence of rare diseases: analysis of the orphanet database</article-title><source>Eur J Hum Genet</source><year>2019</year><volume>28</volume><issue>2</issue><fpage>165</fpage><lpage>73</lpage><pub-id pub-id-type="doi">10.1038/s41431-019-0508-0</pub-id><comment>https://doi.org/10.1038/s41431-019-0508-0</comment></mixed-citation></ref><ref id="CR18"><label>18</label><mixed-citation publication-type="other">Orphadata. <ext-link xlink:href="http://www.orphadata.org/" ext-link-type="url">http://www.orphadata.org/</ext-link>. Accessed 26 June 2021.</mixed-citation></ref><ref id="CR19"><label>19</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lancet</surname><given-names>T</given-names></name></person-group><article-title xml:lang="en">ICD-11</article-title><source>Lancet</source><year>2019</year><volume>393</volume><issue>10188</issue><fpage>2275</fpage><pub-id pub-id-type="doi">10.1016/S0140-6736(19)31205-X</pub-id><comment>https://doi.org/10.1016/s0140-6736(19)31205-x</comment></mixed-citation></ref><ref id="CR20"><label>20</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gkoutos</surname><given-names>GV</given-names></name><name><surname>Schofield</surname><given-names>PN</given-names></name><name><surname>Hoehndorf</surname><given-names>R</given-names></name></person-group><article-title xml:lang="en">The anatomy of phenotype ontologies: principles, properties and applications</article-title><source>Brief Bioinform</source><year>2018</year><volume>19</volume><issue>5</issue><fpage>1008</fpage><lpage>21</lpage><pub-id pub-id-type="doi">10.1093/bib/bbx035</pub-id><comment>https://doi.org/10.1093/bib/bbx035</comment></mixed-citation></ref><ref id="CR21"><label>21</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pendergrass</surname><given-names>SA</given-names></name><name><surname>Brown-Gentry</surname><given-names>K</given-names></name><name><surname>Dudek</surname><given-names>S</given-names></name><name><surname>Frase</surname><given-names>A</given-names></name><name><surname>Torstenson</surname><given-names>ES</given-names></name><name><surname>Goodloe</surname><given-names>R</given-names></name><name><surname>Ambite</surname><given-names>JL</given-names></name><name><surname>Avery</surname><given-names>CL</given-names></name><name><surname>Buyske</surname><given-names>S</given-names></name><name><surname>Bůžková</surname><given-names>P</given-names></name><name><surname>Deelman</surname><given-names>E</given-names></name><name><surname>Fesinmeyer</surname><given-names>MD</given-names></name><name><surname>Haiman</surname><given-names>CA</given-names></name><name><surname>Heiss</surname><given-names>G</given-names></name><name><surname>Hindorff</surname><given-names>LA</given-names></name><name><surname>Hsu</surname><given-names>C-N</given-names></name><name><surname>Jackson</surname><given-names>RD</given-names></name><name><surname>Kooperberg</surname><given-names>C</given-names></name><name><surname>Marchand</surname><given-names>LL</given-names></name><name><surname>Lin</surname><given-names>Y</given-names></name><name><surname>Matise</surname><given-names>TC</given-names></name><name><surname>Monroe</surname><given-names>KR</given-names></name><name><surname>Moreland</surname><given-names>L</given-names></name><name><surname>Park</surname><given-names>SL</given-names></name><name><surname>Reiner</surname><given-names>A</given-names></name><name><surname>Wallace</surname><given-names>R</given-names></name><name><surname>Wilkens</surname><given-names>LR</given-names></name><name><surname>Crawford</surname><given-names>DC</given-names></name><name><surname>Ritchie</surname><given-names>MD</given-names></name></person-group><article-title xml:lang="en">Phenome-wide association study (PheWAS) for detection of pleiotropy within the population architecture using genomics and epidemiology (PAGE) network</article-title><source>PLoS Genet</source><year>2013</year><volume>9</volume><issue>1</issue><fpage>1003087</fpage><pub-id pub-id-type="doi">10.1371/journal.pgen.1003087</pub-id><comment>https://doi.org/10.1371/journal.pgen.1003087</comment></mixed-citation></ref><ref id="CR22"><label>22</label><mixed-citation publication-type="other">Porter HF, O’Reilly PF. Multivariate simulation framework reveals performance of multi-trait GWAS methods. Sci Rep. 2017; 7(1). <ext-link xlink:href="https://doi.org/10.1038/srep38837" ext-link-type="doi">https://doi.org/10.1038/srep38837</ext-link>.</mixed-citation></ref><ref id="CR23"><label>23</label><mixed-citation publication-type="other">Wei W-Q, Denny JC. Extracting research-quality phenotypes from electronic health records to support precision medicine. Genome Med. 2015; 7(1). <ext-link xlink:href="https://doi.org/10.1186/s13073-015-0166-y" ext-link-type="doi">https://doi.org/10.1186/s13073-015-0166-y</ext-link>.</mixed-citation></ref><ref id="CR24"><label>24</label><mixed-citation publication-type="other">Ghiassian SD, Menche J, Chasman DI, Giulianini F, Wang R, Ricchiuto P, Aikawa M, Iwata H, Müller C, Zeller T, Sharma A, Wild P, Lackner K, Singh S, Ridker PM, Blankenberg S, Barabási A-L, Loscalzo J. Endophenotype network models: Common core of complex diseases. Sci Rep. 2016; 6(1). <ext-link xlink:href="https://doi.org/10.1038/srep27414" ext-link-type="doi">https://doi.org/10.1038/srep27414</ext-link>.</mixed-citation></ref><ref id="CR25"><label>25</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Schofield</surname><given-names>PN</given-names></name><name><surname>Gkoutos</surname><given-names>GV</given-names></name><name><surname>Gruenberger</surname><given-names>M</given-names></name><name><surname>Sundberg</surname><given-names>JP</given-names></name><name><surname>Hancock</surname><given-names>JM</given-names></name></person-group><article-title xml:lang="en">Phenotype ontologies for mouse and man: bridging the semantic gap</article-title><source>Dis Model Mech</source><year>2010</year><volume>3</volume><issue>5-6</issue><fpage>281</fpage><lpage>89</lpage><pub-id pub-id-type="doi">10.1242/dmm.002790</pub-id><comment>https://doi.org/10.1242/dmm.002790</comment></mixed-citation></ref><ref id="CR26"><label>26</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Oti</surname><given-names>M</given-names></name><name><surname>Brunner</surname><given-names>HG</given-names></name></person-group><article-title xml:lang="en">The modular nature of genetic diseases</article-title><source>Clin Genet</source><year>2007</year><volume>71</volume><issue>1</issue><fpage>1</fpage><lpage>11</lpage><pub-id pub-id-type="doi">10.1111/j.1399-0004.2006.00708.x</pub-id></mixed-citation></ref><ref id="CR27"><label>27</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Goh</surname><given-names>KI</given-names></name><name><surname>Cusick</surname><given-names>ME</given-names></name><name><surname>Valle</surname><given-names>D</given-names></name><name><surname>Childs</surname><given-names>B</given-names></name><name><surname>Vidal</surname><given-names>M</given-names></name><name><surname>Barabasi</surname><given-names>AL</given-names></name></person-group><article-title xml:lang="en">The human disease network</article-title><source>Proc Natl Acad Sci U S A</source><year>2007</year><volume>104</volume><issue>21</issue><fpage>8685</fpage><lpage>90</lpage><pub-id pub-id-type="doi">10.1073/pnas.0701361104</pub-id></mixed-citation></ref><ref id="CR28"><label>28</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Blair</surname><given-names>DR</given-names></name><name><surname>Lyttle</surname><given-names>CS</given-names></name><name><surname>Mortensen</surname><given-names>JM</given-names></name><name><surname>Bearden</surname><given-names>CF</given-names></name><name><surname>Jensen</surname><given-names>AB</given-names></name><name><surname>Khiabanian</surname><given-names>H</given-names></name><name><surname>Melamed</surname><given-names>R</given-names></name><name><surname>Rabadan</surname><given-names>R</given-names></name><name><surname>Bernstam</surname><given-names>EV</given-names></name><name><surname>Brunak</surname><given-names>S</given-names></name><name><surname>Jensen</surname><given-names>LJ</given-names></name><name><surname>Nicolae</surname><given-names>D</given-names></name><name><surname>Shah</surname><given-names>NH</given-names></name><name><surname>Grossman</surname><given-names>RL</given-names></name><name><surname>Cox</surname><given-names>NJ</given-names></name><name><surname>White</surname><given-names>KP</given-names></name><name><surname>Rzhetsky</surname><given-names>A</given-names></name></person-group><article-title xml:lang="en">A nondegenerate code of deleterious variants in mendelian loci contributes to complex disease risk</article-title><source>Cell</source><year>2013</year><volume>155</volume><issue>1</issue><fpage>70</fpage><lpage>80</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2013.08.030</pub-id><comment>https://doi.org/10.1016/j.cell.2013.08.030</comment></mixed-citation></ref><ref id="CR29"><label>29</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Schriml</surname><given-names>LM</given-names></name><name><surname>Mitraka</surname><given-names>E</given-names></name><name><surname>Munro</surname><given-names>J</given-names></name><name><surname>Tauber</surname><given-names>B</given-names></name><name><surname>Schor</surname><given-names>M</given-names></name><name><surname>Nickle</surname><given-names>L</given-names></name><name><surname>Felix</surname><given-names>V</given-names></name><name><surname>Jeng</surname><given-names>L</given-names></name><name><surname>Bearer</surname><given-names>C</given-names></name><name><surname>Lichenstein</surname><given-names>R</given-names></name><name><surname>Bisordi</surname><given-names>K</given-names></name><name><surname>Campion</surname><given-names>N</given-names></name><name><surname>Hyman</surname><given-names>B</given-names></name><name><surname>Kurland</surname><given-names>D</given-names></name><name><surname>Oates</surname><given-names>CP</given-names></name><name><surname>Kibbey</surname><given-names>S</given-names></name><name><surname>Sreekumar</surname><given-names>P</given-names></name><name><surname>Le</surname><given-names>C</given-names></name><name><surname>Giglio</surname><given-names>M</given-names></name><name><surname>Greene</surname><given-names>C</given-names></name></person-group><article-title xml:lang="en">Human disease ontology 2018 update: classification, content and workflow expansion</article-title><source>Nucleic Acids Res</source><year>2018</year><volume>47</volume><issue>D1</issue><fpage>955</fpage><lpage>62</lpage><pub-id pub-id-type="doi">10.1093/nar/gky1032</pub-id><comment>https://doi.org/10.1093/nar/gky1032</comment></mixed-citation></ref><ref id="CR30"><label>30</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hoehndorf</surname><given-names>R</given-names></name><name><surname>Schofield</surname><given-names>PN</given-names></name><name><surname>Gkoutos</surname><given-names>GV</given-names></name></person-group><article-title xml:lang="en">Analysis of the human diseasome using phenotype similarity between common, genetic, and infectious diseases</article-title><source>Sci Rep</source><year>2015</year><volume>5</volume><fpage>10888</fpage><pub-id pub-id-type="doi">10.1038/srep10888</pub-id><comment>https://doi.org/10.1038/srep10888</comment></mixed-citation></ref><ref id="CR31"><label>31</label><mixed-citation publication-type="other">Human Phenotype Ontology Annotations. <ext-link xlink:href="https://hpo.jax.org/app/download/annotation" ext-link-type="url">https://hpo.jax.org/app/download/annotation</ext-link>. Accessed 19 Apr 2021.</mixed-citation></ref><ref id="CR32"><label>32</label><mixed-citation publication-type="other">Mouse Genome Informatics. <ext-link xlink:href="http://www.informatics.jax.org" ext-link-type="url">http://www.informatics.jax.org</ext-link>. Accessed 19 Apr 2021.</mixed-citation></ref><ref id="CR33"><label>33</label><mixed-citation publication-type="other">ICD, 10. <ext-link xlink:href="https://www.cdc.gov/nchs/icd/icd10cm.htm" ext-link-type="url">https://www.cdc.gov/nchs/icd/icd10cm.htm</ext-link>. Accessed 19 Apr 2021.</mixed-citation></ref><ref id="CR34"><label>34</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Robinson</surname><given-names>PN</given-names></name><name><surname>Köhler</surname><given-names>S</given-names></name><name><surname>Bauer</surname><given-names>S</given-names></name><name><surname>Seelow</surname><given-names>D</given-names></name><name><surname>StefanMundlos</surname><given-names>D</given-names></name></person-group><article-title xml:lang="en">The human phenotype ontology: A tool for annotating and analyzing human hereditary disease</article-title><source>Am J Hum Genet</source><year>2008</year><volume>83</volume><issue>5</issue><fpage>610</fpage><lpage>15</lpage><pub-id pub-id-type="doi">10.1016/j.ajhg.2008.09.017</pub-id></mixed-citation></ref><ref id="CR35"><label>35</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Smith</surname><given-names>CL</given-names></name><name><surname>Eppig</surname><given-names>JT</given-names></name></person-group><article-title xml:lang="en">The mammalian phenotype ontology: enabling robust annotation and comparative analysis</article-title><source>Wiley Interdiscip Rev Syst Biol Med</source><year>2009</year><volume>1</volume><issue>3</issue><fpage>390</fpage><lpage>99</lpage><pub-id pub-id-type="doi">10.1002/wsbm.44</pub-id></mixed-citation></ref><ref id="CR36"><label>36</label><mixed-citation publication-type="other">UMLS. <ext-link xlink:href="https://www.nlm.nih.gov/research/umls/index.html" ext-link-type="url">https://www.nlm.nih.gov/research/umls/index.html</ext-link>. Accessed 19 Apr 2021.</mixed-citation></ref><ref id="CR37"><label>37</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mungall</surname><given-names>CJ</given-names></name><name><surname>McMurry</surname><given-names>JA</given-names></name><name><surname>Köhler</surname><given-names>S</given-names></name><name><surname>Balhoff</surname><given-names>JP</given-names></name><name><surname>Borromeo</surname><given-names>C</given-names></name><name><surname>Brush</surname><given-names>M</given-names></name><name><surname>Carbon</surname><given-names>S</given-names></name><name><surname>Conlin</surname><given-names>T</given-names></name><name><surname>Dunn</surname><given-names>N</given-names></name><name><surname>Engelstad</surname><given-names>M</given-names></name><name><surname>Foster</surname><given-names>E</given-names></name><name><surname>Gourdine</surname><given-names>JP</given-names></name><name><surname>Jacobsen</surname><given-names>JOB</given-names></name><name><surname>Keith</surname><given-names>D</given-names></name><name><surname>Laraway</surname><given-names>B</given-names></name><name><surname>Lewis</surname><given-names>SE</given-names></name><name><surname>NguyenXuan</surname><given-names>J</given-names></name><name><surname>Shefchek</surname><given-names>K</given-names></name><name><surname>Vasilevsky</surname><given-names>N</given-names></name><name><surname>Yuan</surname><given-names>Z</given-names></name><name><surname>Washington</surname><given-names>N</given-names></name><name><surname>Hochheiser</surname><given-names>H</given-names></name><name><surname>Groza</surname><given-names>T</given-names></name><name><surname>Smedley</surname><given-names>D</given-names></name><name><surname>Robinson</surname><given-names>PN</given-names></name><name><surname>Haendel</surname><given-names>MA</given-names></name></person-group><article-title xml:lang="en">The Monarch Initiative: an integrative data and analytic platform connecting phenotypes to genotypes across species</article-title><source>Nucleic Acids Res</source><year>2016</year><volume>45</volume><issue>D1</issue><fpage>712</fpage><lpage>22</lpage><pub-id pub-id-type="doi">10.1093/nar/gkw1128</pub-id><comment>http://dx.doi.org/10.1093/nar/gkw1128. https://academic.oup.com/nar/article-pdf/45/D1/D712/8846933/gkw1128.pdf</comment></mixed-citation></ref><ref id="CR38"><label>38</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hamosh</surname><given-names>A</given-names></name><name><surname>Scott</surname><given-names>A</given-names></name><name><surname>Amberger</surname><given-names>J</given-names></name><name><surname>Valle</surname><given-names>D</given-names></name><name><surname>McKusick</surname><given-names>V</given-names></name></person-group><article-title xml:lang="en">Online mendelian inheritance in man (omim)</article-title><source>Hum Mutat</source><year>2000</year><volume>15</volume><issue>1</issue><fpage>57</fpage><lpage>61</lpage><pub-id pub-id-type="doi">10.1002/(SICI)1098-1004(200001)15:1&lt;57::AID-HUMU12&gt;3.0.CO;2-G</pub-id></mixed-citation></ref><ref id="CR39"><label>39</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Grau</surname><given-names>B</given-names></name><name><surname>Horrocks</surname><given-names>I</given-names></name><name><surname>Motik</surname><given-names>B</given-names></name><name><surname>Parsia</surname><given-names>B</given-names></name><name><surname>Patelschneider</surname><given-names>P</given-names></name><name><surname>Sattler</surname><given-names>U</given-names></name></person-group><article-title xml:lang="en">OWL 2: The next step for OWL</article-title><source>Web Semant Sci Serv Agents World Wide Web</source><year>2008</year><volume>6</volume><issue>4</issue><fpage>309</fpage><lpage>22</lpage><pub-id pub-id-type="doi">10.1016/j.websem.2008.05.001</pub-id></mixed-citation></ref><ref id="CR40"><label>40</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hoehndorf</surname><given-names>R</given-names></name><name><surname>Schofield</surname><given-names>PN</given-names></name><name><surname>Gkoutos</surname><given-names>GV</given-names></name></person-group><article-title xml:lang="en">Phenomenet: a whole-phenome approach to disease gene discovery</article-title><source>Nucleic Acids Res</source><year>2011</year><volume>39</volume><issue>18</issue><fpage>119</fpage><pub-id pub-id-type="doi">10.1093/nar/gkr538</pub-id></mixed-citation></ref><ref id="CR41"><label>41</label><mixed-citation publication-type="other">Wikidata. <ext-link xlink:href="https://www.wikidata.org/wiki/Wikidata:Main_Page" ext-link-type="url">https://www.wikidata.org/wiki/Wikidata:Main_Page</ext-link>. Accessed 19 Apr 2021.</mixed-citation></ref><ref id="CR42"><label>42</label><mixed-citation publication-type="other">MGI, gene-phenotype associations. <ext-link xlink:href="http://www.informatics.jax.org/downloads/reports/MGI_GenePheno.rpt" ext-link-type="url">http://www.informatics.jax.org/downloads/reports/MGI_GenePheno.rpt</ext-link>. Accessed 19 Apr 2021.</mixed-citation></ref><ref id="CR43"><label>43</label><mixed-citation publication-type="other">MGI, gene-disease associations. <ext-link xlink:href="http://www.informatics.jax.org/downloads/reports/MGI_DO.rpt" ext-link-type="url">http://www.informatics.jax.org/downloads/reports/MGI_DO.rpt</ext-link>. Accessed 19 Apr 2021.</mixed-citation></ref><ref id="CR44"><label>44</label><mixed-citation publication-type="other">Apache Lucene. <ext-link xlink:href="https://lucene.apache.org/" ext-link-type="url">https://lucene.apache.org/</ext-link>.Accessed 19 Apr 2021.</mixed-citation></ref><ref id="CR45"><label>45</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Church</surname><given-names>KW</given-names></name><name><surname>Hanks</surname><given-names>P</given-names></name></person-group><article-title xml:lang="en">Word association norms, mutual information and lexicography</article-title><source>Comput Linguist</source><year>1990</year><volume>16</volume><issue>1</issue><fpage>22</fpage><lpage>29</lpage></mixed-citation></ref><ref id="CR46"><label>46</label><mixed-citation publication-type="other">Statpearls. 2021. <ext-link xlink:href="https://www.ncbi.nlm.nih.gov/books/NBK430685/" ext-link-type="url">https://www.ncbi.nlm.nih.gov/books/NBK430685/</ext-link>. Accessed 19 Apr 2021.</mixed-citation></ref><ref id="CR47"><label>47</label><mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Firth</surname><given-names>J</given-names></name><name><surname>Conlon</surname><given-names>C</given-names></name><name><surname>Cox</surname><given-names>T</given-names></name></person-group><source>Oxford Textbook of Medicine</source><year>2020</year><publisher-loc>Oxford</publisher-loc><publisher-name>Oxford University Press</publisher-name><pub-id pub-id-type="doi">10.1093/med/9780198746690.001.0001</pub-id><comment>https://doi.org/10.1093/med/9780198746690.001.0001. https://oxfordmedicine.com/view/10.1093/med/9780198746690.001.0001/med-9780198746690</comment></mixed-citation></ref><ref id="CR48"><label>48</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Vainchenker</surname><given-names>W</given-names></name><name><surname>Constantinescu</surname><given-names>SN</given-names></name><name><surname>Plo</surname><given-names>I</given-names></name></person-group><article-title xml:lang="en">Recent advances in understanding myelofibrosis and essential thrombocythemia</article-title><source>F1000Research</source><year>2016</year><volume>5</volume><fpage>700</fpage><pub-id pub-id-type="doi">10.12688/f1000research.8081.1</pub-id><comment>https://doi.org/10.12688/f1000research.8081.1</comment></mixed-citation></ref><ref id="CR49"><label>49</label><mixed-citation publication-type="other">Chen J, Hu P, Jimenez-Ruiz E, Holter OM, Antonyrajah D, Horrocks I. Owl2vec*: Embedding of owl ontologies. arXiv preprint arXiv:2009.14654. 2020.</mixed-citation></ref><ref id="CR50"><label>50</label><mixed-citation publication-type="other">Mikolov T, Sutskever I, Chen K, Corrado G, Dean J. Distributed representations of words and phrases and their compositionality. arXiv preprint arXiv:1310.4546. 2013.</mixed-citation></ref><ref id="CR51"><label>51</label><mixed-citation publication-type="book"><person-group person-group-type="editor"><name><surname>Dubitzky</surname><given-names>W</given-names></name><name><surname>Wolkenhauer</surname><given-names>O</given-names></name><name><surname>Cho</surname><given-names>K-H</given-names></name><name><surname>Yokota</surname><given-names>H.</given-names></name></person-group><source>Receiver Operating Characteristic (ROC) Curve</source><year>2013</year><publisher-loc>New York</publisher-loc><publisher-name>Springer</publisher-name></mixed-citation></ref><ref id="CR52"><label>52</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Comfere</surname><given-names>NI</given-names></name><name><surname>Macaron</surname><given-names>NC</given-names></name><name><surname>Gibson</surname><given-names>LE</given-names></name></person-group><article-title xml:lang="en">Cutaneous manifestations of wegener?s granulomatosis: a clinicopathologic study of 17 patients and correlation to antineutrophil cytoplasmic antibody status</article-title><source>J Cutan Pathol</source><year>2007</year><volume>34</volume><issue>10</issue><fpage>739</fpage><lpage>47</lpage><pub-id pub-id-type="doi">10.1111/j.1600-0560.2006.00699.x</pub-id><comment>https://doi.org/10.1111/j.1600-0560.2006.00699.x</comment></mixed-citation></ref><ref id="CR53"><label>53</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Giorgi</surname><given-names>D</given-names></name><name><surname>Pace</surname><given-names>F</given-names></name><name><surname>Giorgi</surname><given-names>A</given-names></name><name><surname>Bonomo</surname><given-names>L</given-names></name><name><surname>Gabrieli</surname><given-names>CB</given-names></name></person-group><article-title xml:lang="en">Retinopathy in systemic lupus erythematosus: pathogenesis and approach to therapy</article-title><source>Hum Immunol</source><year>1999</year><volume>60</volume><issue>8</issue><fpage>688</fpage><lpage>96</lpage><pub-id pub-id-type="doi">10.1016/S0198-8859(99)00035-X</pub-id><comment>https://doi.org/10.1016/s0198-8859(99)00035-x</comment></mixed-citation></ref><ref id="CR54"><label>54</label><mixed-citation publication-type="other">Sarntivijai S, Vasant D, Jupp S, Saunders G, Bento AP, Gonzalez D, Betts J, Hasan S, Koscielny G, Dunham I, Parkinson H, Malone J. Linking rare and common disease: mapping clinical disease-phenotypes to ontologies in therapeutic target validation. J Biomed Semant. 2016; 7(1). <ext-link xlink:href="https://doi.org/10.1186/s13326-016-0051-7" ext-link-type="doi">https://doi.org/10.1186/s13326-016-0051-7</ext-link>.</mixed-citation></ref><ref id="CR55"><label>55</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Malone</surname><given-names>J</given-names></name><name><surname>Holloway</surname><given-names>E</given-names></name><name><surname>Adamusiak</surname><given-names>T</given-names></name><name><surname>Kapushesky</surname><given-names>M</given-names></name><name><surname>Zheng</surname><given-names>J</given-names></name><name><surname>Kolesnikov</surname><given-names>N</given-names></name><name><surname>Zhukova</surname><given-names>A</given-names></name><name><surname>Brazma</surname><given-names>A</given-names></name><name><surname>Parkinson</surname><given-names>H</given-names></name></person-group><article-title xml:lang="en">Modeling sample variables with an Experimental Factor Ontology</article-title><source>Bioinformatics</source><year>2010</year><volume>26</volume><issue>8</issue><fpage>1112</fpage><lpage>18</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btq099</pub-id><comment>10.1093/bioinformatics/btq099. https://academic.oup.com/bioinformatics/article-pdf/26/8/1112/13848104/btq099.pdf</comment></mixed-citation></ref></ref-list></ref-list><app-group><app id="App1"><sec id="Sec24"><title>Supplementary Information</title><p><supplementary-material content-type="local-data" id="MOESM1" xlink:title="Supplementary Information"><media xlink:href="MediaObjects/13326_2021_249_MOESM1_ESM.pdf" mimetype="application" mime-subtype="pdf"><caption xml:lang="en"><p><bold>Additional file 1</bold> SPARQL queries. This file contains the three SPARQL queries used to extract ICD-10–phenotype associations and ICD-10–OMIM mappings from Wikidata.</p></caption></media></supplementary-material></p></sec></app></app-group><glossary><title>Abbreviations</title><def-list><def-item><term>DO</term><def><p>Human Disease Ontology</p></def></def-item><def-item><term>EFO</term><def><p>Experimental Factor Ontology</p></def></def-item><def-item><term>HPO</term><def><p>Human Phenotype Ontology</p></def></def-item><def-item><term>IBD</term><def><p>Inflammatory Bowel Disease</p></def></def-item><def-item><term>ICD</term><def><p>International Classification of Diseases</p></def></def-item><def-item><term>MGI</term><def><p>Mouse Genome Informatics</p></def></def-item><def-item><term>MP</term><def><p>Mammalian Phenotype Ontology</p></def></def-item><def-item><term>NPMI</term><def><p>Normalized Pointwise Mutual Information</p></def></def-item><def-item><term>OBAN</term><def><p>Ontology of Biomedical AssociatioN</p></def></def-item><def-item><term>OMIM</term><def><p>Online Mendelian Inheritance in Man</p></def></def-item><def-item><term>OWL</term><def><p>Web Ontology Language</p></def></def-item><def-item><term>ROCAUC</term><def><p>Receiver Operating Characteristic Area Under Curve</p></def></def-item><def-item><term>UKB</term><def><p>UK Biobank</p></def></def-item><def-item><term>UMLS</term><def><p>Unified Medical Language System</p></def></def-item></def-list></glossary><notes notes-type="Misc"><title>Publisher’s Note</title><p>Springer Nature remains neutral with regard to jurisdictional claims in published maps and institutional affiliations.</p></notes></back></article>