<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC " -//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.2 20190208//EN" "JATS-journalpublishing1.dtd">

<article  xml:lang="en" article-type="research-article" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance"  xmlns:xlink="http://www.w3.org/1999/xlink">
  <front>
    <journal-meta>
      <journal-id journal-id-type="publisher-id">bioinformatics</journal-id>
      <journal-title-group>
        <journal-title>Bioinformatics</journal-title>
      </journal-title-group>
      <issn pub-type="ppub">1367-4803</issn>
      <issn pub-type="epub">1367-4811</issn>
      <publisher>
        <publisher-name>Oxford University Press</publisher-name>
      </publisher>
    </journal-meta>
    <article-meta>
      <article-id pub-id-type="doi">10.1093/bioinformatics/btae021</article-id>
      <article-id pub-id-type="publisher-id">btae021</article-id>
      <article-categories>
        <subj-group subj-group-type="category-toc-heading">
          <subject>Original Paper</subject>
          <subj-group subj-group-type="category-toc-heading">
            <subject>Data and text mining</subject>
          </subj-group>
        </subj-group>
        <subj-group subj-group-type="category-taxonomy-collection">
          <subject>AcademicSubjects/SCI01060</subject>
        </subj-group>
      </article-categories>
      <title-group>
        <article-title>Text mining for contexts and relationships in cancer genomics literature</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="author" corresp="yes">
          <contrib-id contrib-id-type="orcid">https://orcid.org/0000-0002-8715-9035</contrib-id>
          <name>
            <surname>Collins</surname>
            <given-names>Charlotte</given-names>
          </name>
          <aff>
            <institution>Language Technology Laboratory, Theoretical and Applied Linguistics, Faculty of Modern and Medieval Languages and Linguistics, University of Cambridge</institution>, Cambridge CB3 9DA, <country country="GB">United Kingdom</country></aff>
          <xref ref-type="corresp" rid="btae021-cor1" />
          <xref ref-type="fn" rid="btae021-FM1" />
          <email xlink:type="simple">cac74@cam.ac.uk</email>
        </contrib>
        <contrib contrib-type="author" corresp="yes">
          <contrib-id contrib-id-type="orcid">https://orcid.org/0000-0002-0998-438X</contrib-id>
          <name>
            <surname>Baker</surname>
            <given-names>Simon</given-names>
          </name>
          <aff>
            <institution>Language Technology Laboratory, Theoretical and Applied Linguistics, Faculty of Modern and Medieval Languages and Linguistics, University of Cambridge</institution>, Cambridge CB3 9DA, <country country="GB">United Kingdom</country></aff>
          <xref ref-type="corresp" rid="btae021-cor1" />
          <email xlink:type="simple">sb895@cam.ac.uk</email>
          <xref ref-type="fn" rid="btae021-FM1" />
        </contrib>
        <contrib contrib-type="author">
          <name>
            <surname>Brown</surname>
            <given-names>Jason</given-names>
          </name>
          <aff>
            <institution>Language Technology Laboratory, Theoretical and Applied Linguistics, Faculty of Modern and Medieval Languages and Linguistics, University of Cambridge</institution>, Cambridge CB3 9DA, <country country="GB">United Kingdom</country></aff>
          <xref ref-type="fn" rid="btae021-FM1" />
        </contrib>
        <contrib contrib-type="author">
          <name>
            <surname>Zheng</surname>
            <given-names>Huiyuan</given-names>
          </name>
          <aff>
            <institution>Institute of Environmental Medicine, Karolinska Institutet</institution>, 171 77 Stockholm, <country country="SE">Sweden</country></aff>
        </contrib>
        <contrib contrib-type="author">
          <name>
            <surname>Chan</surname>
            <given-names>Adelyne</given-names>
          </name>
          <aff>
            <institution>Cancer Research UK Cambridge Institute, Li Ka Shing Centre, University of Cambridge</institution>, Cambridge CB2 0RE, <country country="GB">United Kingdom</country></aff>
        </contrib>
        <contrib contrib-type="author">
          <name>
            <surname>Stenius</surname>
            <given-names>Ulla</given-names>
          </name>
          <aff>
            <institution>Institute of Environmental Medicine, Karolinska Institutet</institution>, 171 77 Stockholm, <country country="SE">Sweden</country></aff>
        </contrib>
        <contrib contrib-type="author">
          <name>
            <surname>Narita</surname>
            <given-names>Masashi</given-names>
          </name>
          <aff>
            <institution>Cancer Research UK Cambridge Institute, Li Ka Shing Centre, University of Cambridge</institution>, Cambridge CB2 0RE, <country country="GB">United Kingdom</country></aff>
        </contrib>
        <contrib contrib-type="author">
          <name>
            <surname>Korhonen</surname>
            <given-names>Anna</given-names>
          </name>
          <aff>
            <institution>Language Technology Laboratory, Theoretical and Applied Linguistics, Faculty of Modern and Medieval Languages and Linguistics, University of Cambridge</institution>, Cambridge CB3 9DA, <country country="GB">United Kingdom</country></aff>
        </contrib>
      </contrib-group>
      <contrib-group>
        <contrib contrib-type="editor">
          <name>
            <surname>Lu</surname>
            <given-names>Zhiyong</given-names>
          </name>
          <role>Associate Editor</role>
        </contrib>
      </contrib-group>
      <author-notes>
        <corresp id="btae021-cor1">Corresponding authors. Language Technology Laboratory, Theoretical and Applied Linguistics, University of Cambridge, Cambridge CB3 9DA, United Kingdom. E-mails: <email>cac74@cam.ac.uk</email> (C.C.); <email>sb895@cam.ac.uk</email> (S.B.)</corresp>
        <fn id="btae021-FM1">
          <p>Equal contribution by Charlotte Collins and Simon Baker.</p>
        </fn>
      </author-notes>
      <pub-date pub-type="cover" iso-8601-date="2024-01-01">
        <month>01</month>
        <year>2024</year>
      </pub-date>
      <pub-date pub-type="collection" iso-8601-date="2024-01-02">
        <day>02</day>
        <month>01</month>
        <year>2024</year>
      </pub-date>
      <pub-date pub-type="epub" iso-8601-date="2024-01-22">
        <day>22</day>
        <month>01</month>
        <year>2024</year>
      </pub-date>
      <volume>40</volume>
      <issue>1</issue>
      <elocation-id>btae021</elocation-id>
      <supplementary-material id="sup1" content-type="data-supplement" mimetype="text" xlink:href="btae021_supplementary_data.pdf">
        <label>btae021_Supplementary_Data</label>
      </supplementary-material>
      <history>
        <date date-type="received">
          <day>13</day>
          <month>04</month>
          <year>2023</year>
        </date>
        <date date-type="rev-recd">
          <day>27</day>
          <month>09</month>
          <year>2023</year>
        </date>
        <date date-type="editorial-decision">
          <day>26</day>
          <month>12</month>
          <year>2023</year>
        </date>
        <date date-type="accepted">
          <day>15</day>
          <month>01</month>
          <year>2024</year>
        </date>
        <date date-type="corrected-typeset">
          <day>27</day>
          <month>01</month>
          <year>2024</year>
        </date>
      </history>
      <permissions>
        <copyright-statement>© The Author(s) 2024. Published by Oxford University Press.</copyright-statement>
        <copyright-year>2024</copyright-year>
        <license license-type="cc-by" xlink:href="https://creativecommons.org/licenses/by/4.0/">
          <license-p>This is an Open Access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted reuse, distribution, and reproduction in any medium, provided the original work is properly cited.</license-p>
        </license>
      </permissions>
      <self-uri xlink:href="btae021.pdf" />
      <abstract abstract-type="abstract">
        <title>Abstract</title>
        <sec id="s1">
          <title>Motivation</title>
          <p>Scientific advances build on the findings of existing research. The 2001 publication of the human genome has led to the production of huge volumes of literature exploring the context-specific functions and interactions of genes. Technology is needed to perform large-scale text mining of research papers to extract the reported actions of genes in specific experimental contexts and cell states, such as cancer, thereby facilitating the design of new therapeutic strategies.</p>
        </sec>
        <sec id="s2">
          <title>Results</title>
          <p>We present a new corpus and Text Mining methodology that can accurately identify and extract the most important details of cancer genomics experiments from biomedical texts. We build a Named Entity Recognition model that accurately extracts relevant experiment details from PubMed abstract text, and a second model that identifies the relationships between them. This system outperforms earlier models and enables the analysis of gene function in diverse and dynamically evolving experimental contexts.</p>
        </sec>
        <sec id="s3">
          <title>Availability and implementation</title>
          <p>Code and data are available here: <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="https://github.com/cambridgeltl/functional-genomics-ie">https://github.com/cambridgeltl/functional-genomics-ie</ext-link>.</p>
        </sec>
      </abstract>
      <funding-group>
        <award-group award-type="grant">
          <funding-source>
            <institution-wrap>
              <institution>UK Research and Innovation</institution>
              <institution-id institution-id-type="DOI">10.13039/100014013</institution-id>
            </institution-wrap>
          </funding-source>
        </award-group>
        <award-group award-type="grant">
          <funding-source>
            <institution-wrap>
              <institution>Amazon Machine Learning Research Award</institution>
            </institution-wrap>
          </funding-source>
        </award-group>
        <award-group award-type="grant">
          <funding-source>
            <institution-wrap>
              <institution>Cancer Research UK Cambridge Institute</institution>
              <institution-id institution-id-type="DOI">10.13039/501100022011</institution-id>
            </institution-wrap>
          </funding-source>
        </award-group>
        <award-group award-type="grant">
          <funding-source>
            <institution-wrap>
              <institution>Biotechnology and Biological Sciences Research Council</institution>
              <institution-id institution-id-type="DOI">10.13039/501100000268</institution-id>
            </institution-wrap>
          </funding-source>
        </award-group>
        <award-group award-type="grant">
          <funding-source>
            <institution-wrap>
              <institution>British Council</institution>
              <institution-id institution-id-type="DOI">10.13039/501100000308</institution-id>
            </institution-wrap>
          </funding-source>
          <award-id>65BX18MNIB</award-id>
        </award-group>
      </funding-group>
      <counts>
        <page-count count="9" />
      </counts>
    </article-meta>
  </front>
  <body>
    <sec>
      <title>1 Introduction</title>
      <p>Scientific advances depend on collective learning, or the ability to communicate, record, and progressively build on knowledge over time. Since the first publication of the human genome in 2001 (<xref ref-type="bibr" rid="btae021-B17">Lander <italic>et al.</italic> 2001</xref>, <xref ref-type="bibr" rid="btae021-B35">Venter <italic>et al.</italic> 2001</xref>), the rapid development of new technologies in the biomedical field of genomics has led to the publication of vast numbers of studies seeking to decode the complex relationship between genotype (genetic constitution) and phenotype (the physical and behavioural attributes resulting from a genotype) (<xref ref-type="bibr" rid="btae021-B25">Przybyla and Gilbert 2022</xref>). The function of individual genes varies with context such that the same gene may perform variably in different species, tissues, disease states, and experimental approaches. Genomics has the potential to revolutionize our understanding and treatment of human diseases, however, the quantity of literature available is too large for researchers to read and analyse manually. Most biomedical research is stored and accessed through the online biomedical database, PubMed, which already contains more than 34 million citations and is continuing to grow exponentially.</p>
      <p>Text mining (TM) can facilitate information extraction from large volumes of literature by turning unstructured, human-generated language into structured data (<xref ref-type="bibr" rid="btae021-B27">Renganathan 2017</xref>). Biomedical TM requires both the means of recognizing biological entities and strategies for identifying contextual details that specify their roles in experiments. TM employs Natural Language Processing (NLP) technology, which trains neural networks of machine-learning algorithms on specialized texts to develop domain-specific language representation models. Recent developments in NLP, such as Transformers (<xref ref-type="bibr" rid="btae021-B18">Lee <italic>et al.</italic> 2020</xref>), are used for general biomedical TM tasks. Here, we seek to develop the first TM methodology that is specifically customized for extracting the contexts of genomics experiments. We develop a Named Entity Recognition (NER) model as well as a relation extraction model to allow the accurate identification and extraction of methods of gene perturbation, phenotypes, and important details of experiment context such as species, cell type, and experiment setting (<italic>in vitro</italic> or <italic>in vivo</italic>). Previous work has focused on extracting genotypes and associated phenotypes (<xref ref-type="bibr" rid="btae021-B37">Xing <italic>et al.</italic> 2018</xref>, <xref ref-type="bibr" rid="btae021-B30">Sousa <italic>et al.</italic> 2019</xref>) from scientific literature without linking them to associated contexts, thereby missing key information.</p>
      <p>We focus on the functional genomics of cancer, a domain in which TM could have particularly high impact. Cancer is one of the leading causes of death worldwide, with 19.3 million new cases and 10 million deaths estimated to occur in 2020 alone (<xref ref-type="bibr" rid="btae021-B33">Sung <italic>et al.</italic> 2021</xref>). It is therefore the subject of intensive research efforts; as of February 2023, a search of the PubMed biomedical database for the term ‘cancer’ returned some 4 795 969 citations. Cancer genomics explores how DNA sequences and gene expression patterns differ between healthy cells and cancer cells. Genetic differences define the distinct types of cancers and cancer cell subpopulations and also dictate their different susceptibilities to pharmacogenetic intervention (<xref ref-type="bibr" rid="btae021-B5">Berger and Mardis 2018</xref>). A key focus of cancer genomics research is the identification of genes which regulate cell death in different cell types and contexts. By pinpointing those genes that regulate cell death in cancer cells, but not in healthy cells, it may be possible to develop new drugs that can precisely target cancer cells whilst sparing healthy cells from harmful toxic effects. Technology that can automatically extract and collate contextualized genomics data from the entire PubMed corpus could help scientists identify the most promising new gene targets and thereby accelerate the development of cancer therapeutics.</p>
    </sec>
    <sec>
      <title>2 Background</title>
      <sec>
        <title>2.1 Cancer genomics and cell death</title>
        <p>Understanding the genetics of cell death is a major focus of the cancer genomics field. The balance between the rate of cell division and the rate of cell death forms the essential mechanism by which multicellular organisms maintain the correct number of cells in their bodies. Failure to properly regulate this process can lead to uncontrolled cell proliferation and cancer.</p>
        <p>Cell death can occur either as a result of external assault or through a process of ‘cell suicide’ in which genetically regulated internal programmes drive the active pursuit of cell death in order to clear damaged or unwanted cells from the body [reviewed in <xref ref-type="bibr" rid="btae021-B31">Strasser and Vaux (2020)</xref>]. Cell death results from the failure of essential metabolic functions [reviewed in <xref ref-type="bibr" rid="btae021-B11">Galluzzi <italic>et al.</italic> (2018)</xref>] and can occur through different mechanisms distinguishable by distinct morphological and genetic characteristics. Some commonly described cell death types are ‘Apoptosis’ (<xref ref-type="bibr" rid="btae021-B14">Kerr <italic>et al.</italic> 1972</xref>), a regulated form of cell death that is characterized by cytoplasmic shrinkage, fragmentation of nuclei, chromatin condensation, and membrane blebbing, ‘Autophagy’ or ‘self-eating’ (<xref ref-type="bibr" rid="btae021-B21">Ohsumi 2014</xref>), which is a an alternative form of cell death in some contexts and is characterized by the development of double-membraned autophagosomes (<xref ref-type="bibr" rid="btae021-B21">Ohsumi 2014</xref>), and ‘Necrosis’, which can result from either external trauma or internal activation and lacks the distinctive morphological features of apoptosis and autophagy [reviewed in <xref ref-type="bibr" rid="btae021-B11">Galluzzi <italic>et al.</italic> (2018)</xref>]. Other cell death subtypes include distinct forms specific to different cell types and contexts. Of particular significance for cancer research, the different mechanisms of cell death are regulated by different groups of genes, described in <xref ref-type="supplementary-material" rid="sup1">Supplementary Table S1</xref>.</p>
        <p>The different forms of cell death have multiple roles in the pathogenesis and treatment of cancers. Cancer cells are characterized by their abilities to evade the normal mechanisms of cell death and to proliferate in an uncontrolled way, resulting in tumours. In addition, death of pre-cancerous cells within a tumour can lead to the release of mitogenic factors, which induce compensatory proliferation in adjacent cells (<xref ref-type="bibr" rid="btae021-B16">Labi and Erlacher 2015</xref>, <xref ref-type="bibr" rid="btae021-B6">Celis <italic>et al.</italic> 2022</xref>). Cancer treatments include the use of radiation or drugs to induce death of cancer cells, whilst aiming to minimize harmful side-effects. The ability of subsets of tumour cells to resist cell death, however, can lead to the re-population of tumours with more aggressive clones and subsequent cancer progression. In addition, a subset of surviving tumour cells may adopt a senescence phenotype and subsequently promote wide-ranging adverse effects including inflammation, cardiac dysfunction, and cancer recurrence (<xref ref-type="bibr" rid="btae021-B9">Demaria <italic>et al.</italic> 2017</xref>). It is therefore of great interest to develop drugs that can target specific genes to kill therapeutically significant subsets of cancer cells without toxicity to normal, healthy cells. An expansive body of research has aimed to generate a detailed understanding of the types of cell death that occur in cancer and non-cancer cell types and of the different groups of genes that regulate them. To maximize the value of this literature to the cancer research and genomics communities, we need to develop tailored TM methodology that can carry out large-scale extraction of the actions of genes in different contexts and cell death subtypes. The data generated could identify novel genes that regulate cell death specifically in cancer cells, thereby allowing the development of targeted cancer drugs and opening up new therapeutic avenues.</p>
      </sec>
      <sec>
        <title>2.2 Biomedical TM</title>
        <p>TM is a specialized field in the biomedical domain and has been used to address many different challenges in biology, cancer research, and drug discovery. Several methods and web-based resources for the analysis of genomics data have been developed, including LitMiner (<xref ref-type="bibr" rid="btae021-B8">Demaine <italic>et al.</italic> 2006</xref>), GeneMANIA (<xref ref-type="bibr" rid="btae021-B36">Warde-Farley <italic>et al.</italic> 2010</xref>), OnTheFly (<xref ref-type="bibr" rid="btae021-B23">Pafilis <italic>et al.</italic> 2013</xref>), SIGNOR (<xref ref-type="bibr" rid="btae021-B24">Perfetto <italic>et al.</italic> 2016</xref>), Enrchr (<xref ref-type="bibr" rid="btae021-B15">Kuleshov <italic>et al.</italic> 2016</xref>), Cancer Hallmarks Analytics Tool (<xref ref-type="bibr" rid="btae021-B4">Baker <italic>et al.</italic> 2016</xref>, <xref ref-type="bibr" rid="btae021-B3">2017</xref>), LION-LBD (<xref ref-type="bibr" rid="btae021-B26">Pyysalo <italic>et al.</italic> 2019</xref>), GENETEX (<xref ref-type="bibr" rid="btae021-B20">Miller and Shalhout 2021</xref>), Cancer Dependency Map (<xref ref-type="bibr" rid="btae021-B29">Shimada <italic>et al.</italic> 2021</xref>), and the Database of Essential Genes (DEG 15) (<xref ref-type="bibr" rid="btae021-B19">Luo <italic>et al.</italic> 2021</xref>). While useful for many purposes, none of these existing resources are specifically adapted for extraction of functional genomics data from journal papers. They are limited in their ability to extract context from semantically rich texts, searches can be performed for gene(s) but not for phenotypes or details of experiment context, and their results are infrequently updated and therefore quickly become out-of-date in the fast moving genomics field.</p>
        <p>Biomedical TM aims to both recognize biological entities and to capture the ways in which two or more entities interact, in relationships commonly termed ‘bio-events’(<xref ref-type="bibr" rid="btae021-B1">Ananiadou <italic>et al.</italic> 2015</xref>). To extract the most important details of genomics experiments, it is necessary to identify gene names and symbols, the methods by which genes are experimentally perturbed, the phenotypes that result from gene perturbations and the contexts in which experiments are carried out. Resources are available for the automatic recognition and grounding of genes, such as the Gene Ontology database (<xref ref-type="bibr" rid="btae021-B2">Ashburner <italic>et al.</italic> 2000</xref>), and of phenotypes, such as the Human Phenotype Ontology (HPO) (<xref ref-type="bibr" rid="btae021-B13">Groza <italic>et al.</italic> 2015</xref>), but there are no equivalent resources to aid recognition of important contextual details, such as methods of gene perturbation, species, cell or tissue type, and experimental approach. Of note, the names of cell lines and gene constructs are often highly individualized and may appear in only a small number, or in some cases a single paper(s). Further, the co-occurrence of two or more entities in the same sentence is not always evidence of their interaction (<xref ref-type="bibr" rid="btae021-B7">Chun <italic>et al.</italic> 2006</xref>), and some entities may interact with multiple other entities which could be either adjacent or distant.</p>
        <p>We aim here to address the significant challenges of accurately identifying relevant cancer genomics entities in large volumes of literature and of extracting these entities together with their complex inter-relationships. To this end, we develop a new corpus of biomedical literature that has been labelled with both genes and phenotypes and important details of experiment context. We apply the latest machine-learning algorithms, such as Transformers (<xref ref-type="bibr" rid="btae021-B10">Devlin <italic>et al.</italic> 2018</xref>) and BioBERT (<xref ref-type="bibr" rid="btae021-B18">Lee <italic>et al.</italic> 2020</xref>), to construct two transformer models that are specifically targeted towards identifying cancer functional genomics entities and linking these entities to relevant context identified in text. The novel ability to extract context represents a significant advance in biomedical TM methodology and extends researchers’ ability to obtain the most highly relevant genomics information from literature.</p>
      </sec>
    </sec>
    <sec>
      <title>3 Materials and methods</title>
      <sec>
        <title>3.1 Corpus annotation</title>
        <sec>
          <title>3.1.1 PubMed literature retrieval</title>
          <p>To obtain relevant texts, we retrieved abstracts from eight PubMed-listed journals that are prominent within the cancer or cell death fields; these were ‘Apoptosis’, ‘Autophagy’, ‘Cancer Cell’, ‘Cancer Research’, ‘Cell’, ‘Cell Death and Differentiation’, ‘Cell Death and Disease’, and ‘Genes and Development’. Using the PubMed Advanced Search tool, we selected abstracts that included one of these titles as a journal name but excluded ‘review’ as a publication type. The first 100 abstracts (listed by PubMed on 12 May 2021) for each journal were downloaded. Abstracts were manually processed into a total of 800 individual text files, in which only the article title and the main body of the abstract were included, with the author list and all other extraneous text being removed.</p>
        </sec>
        <sec>
          <title>3.1.2 Cell death, cancer, and genetics terminology</title>
          <p>To systematically extract the key elements of experiment descriptions from texts, we focused on words and phrases defined by four main categories. These were chosen through discussion with experts in the cancer and cell death research field and capture the most relevant details of cancer and genomics experiments. The four categories were: ‘Perturbing actions’, ‘Contexts’, ‘Phenotypes’, and ‘Effects’. (i) Perturbing actions are defined experimental manipulations that relate to the regulation of any named gene or genes. We included gene names, symbols, and recognized aliases, using <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="http://www.genecards.org">www.genecards.org</ext-link> as a reference. (ii) Contexts are the experimental models subjected to a Perturbing action, e.g. human patients, laboratory animals or cultured primary cells and cell lines. (iii) Phenotypes are the induced changes in the physical characteristics and/or behaviour of the cell or organism that result from a Perturbing action. We restricted this category to terms specific to cell death, terms specific to cancer development and to some common cell biology terms, which can relate to the behaviour of cells in both cancerous and healthy tissues. The list of terms was chosen with reference to relevant literature (<xref ref-type="bibr" rid="btae021-B11">Galluzzi <italic>et al.</italic> 2018</xref>) and after discussion with experts from the cell death and cancer field at Cancer Research UK Cambridge Research Institute, UK. (iv) Effects are the direction in which a Phenotype is regulated, i.e. positively or negatively. A full list of markable words and phrases is shown in <xref ref-type="table" rid="btae021-T1">Table 1</xref>.</p>
          <table-wrap id="btae021-T1">
            <label>Table 1.</label>
            <caption>
              <p>Definitions of each entity category and corresponding assertions for each category.</p>
            </caption>
            <table frame="hsides" rules="groups">
              <colgroup>
                <col valign="top" align="left" />
                <col valign="top" align="left" />
                <col valign="top" align="left" />
              </colgroup>
              <thead>
                <tr>
                  <th>Entity type</th>
                  <th>Definition</th>
                  <th>Assertions used to label marked entities</th>
                </tr>
              </thead>
              <tbody>
                <tr>
                  <td>Perturbing action</td>
                  <td>Defined experimental manipulations that relate to the regulation of a named gene or genes.</td>
                  <td>Gene loss-of-function, gene gain-of-function, RNAi/knockdown, pharmacological inhibition, pharmacological augmentation, other.</td>
                </tr>
                <tr>
                  <td>Context</td>
                  <td>The cells or organisms that are the subject of an experimental manipulation.</td>
                  <td>Patient, organism, tissue/organ, neoplasm, graft, xenograft, cells, transformed cells, organoid, <italic>in vitro</italic>, <italic>in vivo</italic>.</td>
                </tr>
                <tr>
                  <td>Phenotype</td>
                  <td>The induced changes in the physical characteristics and/or behaviour of the organism or cell.</td>
                  <td>Cell death terms: apoptosis, anoikis, autophagy, cell death, entosis, ferroptosis, mitophagy, necroptosis, necrosis, oncosis, and pyroptosis. Cancer terms: metastasis, transformation, tumour-growth, tumourigenesis, tumour initiation, tumour progression, and tumour regression. General cell biology terms: adhesion, cell cycle arrest, cell growth, cell survival, colony formation, differentiation, epithelial-mesenchymal transition, invasion, migration, proliferation, quiescence, self-renewal, and senescence.</td>
                </tr>
                <tr>
                  <td>Effect</td>
                  <td>If a phenotype is regulated positively or negatively.</td>
                  <td>Positive, negative, regulates, rescues, no effect.</td>
                </tr>
              </tbody>
            </table>
          </table-wrap>
        </sec>
        <sec>
          <title>3.1.3 Annotation of texts</title>
          <p>The annotation was carried out by a biologist with 9 years of postgraduate experience in cell biology experimental research. The Multi-Annotation Environment (MAE) annotation tool was used (<xref ref-type="bibr" rid="btae021-B32">Stubbs 2011</xref>, <xref ref-type="bibr" rid="btae021-B28">Rim 2016</xref>) (<ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="http://keighrim.github.io/mae-annotation/">http://keighrim.github.io/mae-annotation/</ext-link>). The framework of the annotation task was defined using a custom DTD file.</p>
          <p>Abstracts were annotated in list order following retrieval from PubMed. Only abstracts linked to primary research articles were annotated. Reviews, commentaries, and editorial articles were excluded from the task. Annotation was carried out at whole text level; however, only the parts of abstracts containing descriptions of experiments were marked. We did not include text that formed part of the introduction or discussion, nor did we mark the text of purely descriptive studies or parts of studies, such as accounts of gene expression patterns.</p>
          <p>Annotation was carried out at the level of whole abstracts. Words or continuous phrases were marked when they both (i) formed part of the description of an experiment and (ii) corresponded to one of the terms of interest listed in <xref ref-type="table" rid="btae021-T2">Table 2</xref>. Descriptions of experiments (though not individual entities) sometimes spanned multiple sentences. All marked Perturbing actions included a gene name or symbol, and marked Contexts included names of cell types, cell lines, species, and breeding lines when present. In some abstracts, words or phrases from all four categories were marked; in other abstracts it was only possible to mark words or phrases from some of the categories. The total number of marked entities in the corpus was 10 458.</p>
          <table-wrap id="btae021-T2">
            <label>Table 2.</label>
            <caption>
              <p>An example of marking and labelling the description of an experiment.<xref ref-type="table-fn" rid="tblfn1"><sup>a</sup></xref></p>
            </caption>
            <table frame="hsides" rules="groups">
              <colgroup>
                <col valign="top" align="left" />
                <col valign="top" align="left" />
                <col valign="top" align="left" />
              </colgroup>
              <thead>
                <tr>
                  <th>Marked text segment</th>
                  <th>Category</th>
                  <th>Assertion</th>
                </tr>
              </thead>
              <tbody>
                <tr>
                  <td>Silencing Lnc-EPIC1 by siRNA</td>
                  <td>Perturbing action</td>
                  <td>RNAi/knockdown</td>
                </tr>
                <tr>
                  <td>Inhibit</td>
                  <td>Effect</td>
                  <td>Negative</td>
                </tr>
                <tr>
                  <td>Cell growth</td>
                  <td>Phenotype</td>
                  <td>Cell growth</td>
                </tr>
                <tr>
                  <td>Colony formation</td>
                  <td>Phenotype</td>
                  <td>Colony formation</td>
                </tr>
                <tr>
                  <td>PC cells</td>
                  <td>Context</td>
                  <td>Cells</td>
                </tr>
                <tr>
                  <td>Induced</td>
                  <td>Effect</td>
                  <td>Positive</td>
                </tr>
                <tr>
                  <td>G1/S cell cycle arrest</td>
                  <td>Phenotype</td>
                  <td>Cell cycle arrest</td>
                </tr>
                <tr>
                  <td>Apoptosis</td>
                  <td>Phenotype</td>
                  <td>Apoptosis</td>
                </tr>
                <tr>
                  <td>PC cells</td>
                  <td>Context</td>
                  <td>Cells</td>
                </tr>
              </tbody>
            </table>
            <table-wrap-foot>
              <fn id="tblfn1">
                <label>a</label>
                <p>Silencing Lnc-EPIC1 by siRNA targeting could significantly inhibit the cell growth and colony formation ability of PC cells and induced G1/S cell cycle arrest and apoptosis in PC cells.</p>
              </fn>
            </table-wrap-foot>
          </table-wrap>
          <p>The MAE tool allows the addition of labels, or ‘assertions’, to marked entities. Lists of assertions relating to each category of markable entity (Perturbing actions, Contexts, Phenotypes, and Effects) were defined within the DTD document and appeared as drop-down lists during the annotation process. Each marked entity was labelled with the appropriate assertion to identify its meaning. For example, ‘cell’, ‘myoblast’, and ‘keratinocyte’ were each labelled with the assertion ‘cell’, as all these terms describe cell types; ‘Rag2 -/-’ was labelled with the assertion ‘gene loss-of-function’, as ‘-/-’ denotes knockout of the Rag2 gene in both alleles. An example of marking and labelling the text of an experiment description is shown in <xref ref-type="table" rid="btae021-T2">Table 2</xref>.</p>
        </sec>
        <sec>
          <title>3.1.4 Linking marked entities</title>
          <p>Many abstracts contained multiple examples of Perturbing actions, Contexts, Phenotypes, and Effects. We created Link tags to record links between four marked entities (one from each category) that formed part of the description of the same experiment. Because the descriptions of experiments sometimes spanned more than one sentence, the four entities in each linked group were sometimes derived from more than one sentence. Where a description of one experiment generated multiple marked entities within one or more categories, we created multiple Link tags. An example of creating Link tags between marked entities is shown in <xref ref-type="fig" rid="btae021-F1">Fig. 1</xref>. The total number of entities that were both marked and linked was 4697.</p>
          <fig id="btae021-F1">
            <label>Figure 1.</label>
            <caption>
              <p>An example of marking and linking entities. Marked entities (shaded) are assigned to one of four categories: Perturbing actions i.e. Silencing Lnc-EPIC1 by siRNA targeting (purple), Contexts i.e. PC cells (green), Effects i.e. inhibit; induce (yellow) or Phenotypes i.e. cell growth; colony formation; G1/S cell cycle arrest; apoptosis (blue). There are four groups (1, 2, 3, and 4) of four linked (related) entities. Each linked group contains one entity from each category, labelled with the same number. Some entities participate in more than one group and are therefore labelled with more than one number.</p>
            </caption>
            <graphic xmlns:xlink="http://www.w3.org/1999/xlink" mimetype="image" xlink:href="btae021f1.jpeg" />
          </fig>
        </sec>
        <sec>
          <title>3.1.5 Statistics of annotated data</title>
          <p>To measure the precision of the annotation process, a secondary annotator with cell biology experience undertook annotation of a subset of 30 abstracts (numbers 561–590). The results of the primary annotator and the secondary annotator were compared using a statistical tool within the MAE program to calculate values for Krippendorf’s Alpha-U, where a value of one indicates complete agreement. Good text marking (the correct identification of markable entities) agreement was achieved with a mean of 0.79 across all four categories (with individual values for Perturbing actions of 0.72, for Contexts of 0.82, for Phenotypes of 0.86, and for Effects of 0.79). Disagreement between annotators most often arose from the omission of whole entities rather than variation in entity boundaries, though examples of both occurred. We also achieved generally good agreement for labelling (attaching correct assertions) of marked entities within all four categories, with Krippendorf’s Alpha-U values for Perturbing actions of 0.71, for Contexts of 0.74, for Phenotypes of 0.84, and for Effects of 0.68. For both text marking and entity labelling, the greatest precision was found within the Phenotypes category.</p>
        </sec>
      </sec>
      <sec>
        <title>3.2 Natural Language Processing</title>
        <sec>
          <title>3.2.1 Data pre-processing</title>
          <p>The MAE tool XML output was converted to JSON via the python standard library. This results in the text of each abstract accompanied by lists of tags for each category. Each tag has a unique id and contains the span of text and label. There are also lists of linked entities, each with a unique id, the tag ids, and text spans for the tags. The per-tag data were converted to per-word data for each word in each abstract. Each datum contained the word, the tokenization of that word, and any tags it had. Each tag entry contained the tag id, all link ids relating to that tag, the schema label, and the tag label itself. We used the <bold>B</bold>eginning <bold>I</bold>nside, <bold>O</bold>utside (BIO) format for tagging—words at the start of a tagged span had the ‘B’ schema label for that tag with ‘I’ for subsequent tags, and an ‘O’ to denote tokens that were not tagged.</p>
          <p>For each word, its tags were copied onto each token, and then each token was converted to an id by the tokenizer before being input into the model. Label outputs were represented as one-hot encoded vectors, with one vector for each category of label and an additional one whose elements were the tag categories themselves. For each label the ‘B’ and ‘I’ versions had separate elements in the one-hot representation. Links were represented as an upper triangular matrix with element <italic>ij</italic> (where <inline-formula id="IE1"><mml:math id="IM1" xmlns:mml="http://www.w3.org/1998/Math/MathML" display="inline"><mml:mrow><mml:mi>i</mml:mi><mml:mo>≤</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:math></inline-formula>) being one if the <italic>i</italic>th and <italic>j</italic>th tokens of the abstract were linked in any tag, and zero if not.</p>
        </sec>
        <sec>
          <title>3.2.2 Machine learning</title>
          <p>We used two neural network models. The first (<xref ref-type="fig" rid="btae021-F2">Fig. 2</xref>) is a NER model that tags spans of text with one of the four category labels (Perturbing Action, Context, Effect, and Phenotype), as well as assigns a relevant assertion listed in <xref ref-type="table" rid="btae021-T1">Table 1</xref>. The second model (<xref ref-type="fig" rid="btae021-F3">Fig. 3</xref>) links the relevant entities identified by the NER model. For example: a perturbing action can be linked to an effect, a phenotype etc., as illustrated in the example shown in <xref ref-type="fig" rid="btae021-F1">Fig. 1</xref>.</p>
          <fig id="btae021-F2">
            <label>Figure 2.</label>
            <caption>
              <p>Architecture of the tag prediction neural model. A transformer produces an embedding of each token, which is fed into several different predictors. Each predictor is a linear classifier. The Category Predictor determines which of the four categories (Perturbing action, Context, Effect, and Phenotype) a token is labelled, while the four specific category predictors predict an assertion term under the given category, see <xref ref-type="table" rid="btae021-T1 btae021-T2">Tables 1 and 2</xref>.</p>
            </caption>
            <graphic xmlns:xlink="http://www.w3.org/1999/xlink" mimetype="image" xlink:href="btae021f2.jpeg"></graphic>
          </fig>
          <fig id="btae021-F3">
            <label>Figure 3.</label>
            <caption>
              <p>The link prediction model makes a binary prediction whether two entities that appear in the text should be associated with each other. The model is composed of two predictors (linear classifiers): the Link Predictor determines if two tokens are part of tags, which are linked, whilst the same-tag predictor determines if two tokens are a part of the same entity tag. Each predictor takes a concatenated pair of embeddings from the transformer as input.</p>
            </caption>
            <graphic xmlns:xlink="http://www.w3.org/1999/xlink" mimetype="image" xlink:href="btae021f3.jpeg"></graphic>
          </fig>
          <p>Both models consist of two parts. The first is a fine-tuned pre-trained BioBERT transformer (<xref ref-type="bibr" rid="btae021-B18">Lee <italic>et al.</italic> 2020</xref>) and the second is a set of predictors. Each predictor is a linear classifier—a dense layer followed by a softmax non-linearity. The category predictor decides which of the four main labels to assign to the token (or predict a non-assignment); likewise; for each of the categories there is a specific head to predict an assertion for that category. The heads were applied to each of the token embeddings produced by the transformer to predict the label for that token. We used cross-entropy loss function for training.</p>
          <p>It was found that the inclusion of a category predictor improved training by increasing the effective amount of training data for the transformer.</p>
          <p>The linking model has two predictors, one for whether two tokens are part of tags, which are linked, and one for whether or not they are part of the same tag. Since we are using a BIO schema for our tags, this latter predictor is not required since this information is encoded into the schema. Thus, results on its performance will not be reported. However, it was found that the inclusion of this second predictor aided the fine-tuning of the transformer, thus improving the accuracy of the first predictor, by increasing the effective amount of training data for the transformer fine-tuning. During the training process we concatenated every possible pair of token embedding vectors. The model predicts if tokens were linked via binary classification. We trained the model using a binary cross-entropy objective.</p>
          <p>For both models, we allowed the transformer model’s parameters to be updated during the training process. As the transformers operate on the token-level they produce token-level classifications. The MAE tool operates on the word level, so outputs on tokens were converted to outputs on words by applying the prediction from the first token of the word. Other methods, such as averaging the predictions or taking a maximum prediction across all tokens, gave similar results so the simpler method was favoured.</p>
          <p>We experimented with different baseline pre-trained transformer models and found that BioBERT to be the strongest for both the tagging and linking tasks. We also considered having each predictor with its own transformer base but early experiments showed this to be less effective as there has little training data for each specific category. We tried this again with our final architecture and hyperparameters and found it to perform similarly. Having multiple predictors on one transformer also reduces the number of transformers needing to be fine-tuned, thus training far quicker.</p>
          <p>In order to mitigate the effects of having a small number of positive examples in the data, loss balancing was employed in both models, where losses for infrequent tags were given a higher weighting; this includes the ‘None’ tag label.</p>
        </sec>
      </sec>
    </sec>
    <sec>
      <title>4 Results</title>
      <sec>
        <title>4.1 Intrinsic evaluation</title>
        <p>The data were split randomly to get a 10% size holdout final testing set. The remaining data were used in <italic>k</italic>-fold (<italic>k</italic> = 4) cross-validation, with hyperparameters being optimized for validation score. Due to the amount of data, the testing set was small so both testing and cross fold validation accuracy will be reported to give a clear indication of performance.</p>
        <p>The linking model was trained, validated, and tested using the ground truth tags to avoid tag errors affecting the evaluation of the linking model’s accuracy. We measured the performance using standard metrics: precision, recall, and <inline-formula id="IE2"><mml:math id="IM2" xmlns:mml="http://www.w3.org/1998/Math/MathML" display="inline"><mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mn>1</mml:mn></mml:msub></mml:mrow></mml:mrow></mml:math></inline-formula> score. We averaged the scores across our cross-validation folds. We determined a true positive (TP) prediction when the tagging model correctly predicts the true label. A false positive (FP) is an incorrect prediction that is not the ‘None’ label. A False Negative was an incorrect prediction when the true label was not the ‘None’ label. If the prediction and true label mismatched and neither were the ‘None’ label, it was counted as both a FP and false negative. This overlap was properly dealt with when calculating metrics. For linking each possible link between tagged tokens was considered. TPs were predicted when they actually linked, FPs were predicted but not actually linked, and false negatives were not predicted but were actually linked.</p>
        <p>
          <xref ref-type="table" rid="btae021-T3">Table 3</xref> summarizes the models performance for each specific category using the mean of the <italic>k</italic>-fold validation. The overall average <inline-formula id="IE3"><mml:math id="IM3" xmlns:mml="http://www.w3.org/1998/Math/MathML" display="inline"><mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mn>1</mml:mn></mml:msub></mml:mrow></mml:mrow></mml:math></inline-formula> score across the predictors is 0.76. The individual <inline-formula id="IE4"><mml:math id="IM4" xmlns:mml="http://www.w3.org/1998/Math/MathML" display="inline"><mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mn>1</mml:mn></mml:msub></mml:mrow></mml:mrow></mml:math></inline-formula> scores for each predictor head were: Category: 0.80, Perturbing action: 0.76, Context: 0.74, Effect: 0.74, and Phenotype: 0.78. For the held-out test dataset, the average <inline-formula id="IE5"><mml:math id="IM5" xmlns:mml="http://www.w3.org/1998/Math/MathML" display="inline"><mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mn>1</mml:mn></mml:msub></mml:mrow></mml:mrow></mml:math></inline-formula> score for the predictors was 0.78, and for each head were: Category: 0.80, Perturbing action: 0.78, Context: 0.70, Effect: 0.80, and Phenotype: 0.82. Due to the small test set size, the consistent improvement of these scores over the validation ones is most likely due to the test set abstracts randomly being easier to tag.</p>
        <table-wrap id="btae021-T3">
          <label>Table 3.</label>
          <caption>
            <p>Tagging model performance per specific category using <italic>k</italic>-fold cross-validation average, with <inline-formula id="IE6"><mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM6" display="inline"><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>4</mml:mn></mml:mrow></mml:math></inline-formula>.</p>
          </caption>
          <table frame="hsides" rules="groups">
            <colgroup>
              <col valign="top" align="left" />
              <col valign="top" align="char" char="." />
              <col valign="top" align="char" char="." />
              <col valign="top" align="char" char="." />
            </colgroup>
            <thead>
              <tr>
                <th>Category</th>
                <th>Precision</th>
                <th>Recall</th>
                <th>
                  <inline-formula id="IE7">
                    <mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM7" display="inline">
                      <mml:mrow>
                        <mml:mrow>
                          <mml:msub>
                            <mml:mrow>
                              <mml:mi>F</mml:mi>
                            </mml:mrow>
                            <mml:mn>1</mml:mn>
                          </mml:msub>
                        </mml:mrow>
                      </mml:mrow>
                    </mml:math>
                  </inline-formula> score</th>
              </tr>
            </thead>
            <tbody>
              <tr>
                <td>Perturbing action</td>
                <td>0.743</td>
                <td>0.786</td>
                <td>0.764</td>
              </tr>
              <tr>
                <td>Context</td>
                <td>0.725</td>
                <td>0.765</td>
                <td>0.744</td>
              </tr>
              <tr>
                <td>Effect</td>
                <td>0.741</td>
                <td>0.740</td>
                <td>0.740</td>
              </tr>
              <tr>
                <td>Phenotype</td>
                <td>0.760</td>
                <td>0.791</td>
                <td>0.775</td>
              </tr>
            </tbody>
          </table>
        </table-wrap>
        <p>We also evaluated our linking model. For the <italic>k</italic>-fold validation the links between tags were predicted with a precision of 0.85, recall of 0.96, and <inline-formula id="IE8"><mml:math id="IM8" xmlns:mml="http://www.w3.org/1998/Math/MathML" display="inline"><mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mn>1</mml:mn></mml:msub></mml:mrow></mml:mrow></mml:math></inline-formula> score of 0.90. For the test set it achieved a precision of 0.95, recall of 0.94, and <inline-formula id="IE9"><mml:math id="IM9" xmlns:mml="http://www.w3.org/1998/Math/MathML" display="inline"><mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mn>1</mml:mn></mml:msub></mml:mrow></mml:mrow></mml:math></inline-formula> score of 0.94. For reference, using a co-occurrence baseline (i.e. where all entities that mentioned in the same sentence are assumed to be related) would result in an <inline-formula id="IE10"><mml:math id="IM10" xmlns:mml="http://www.w3.org/1998/Math/MathML" display="inline"><mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mn>1</mml:mn></mml:msub></mml:mrow></mml:mrow></mml:math></inline-formula> score of 0.42 due to the high number of FPs.</p>
        <p>The results for the ablation study testing different pre-trained transformers baselines, and having each predictor with its own transformer, is given in <xref ref-type="table" rid="btae021-T4">Table 4</xref>. For brevity only the <inline-formula id="IE11"><mml:math id="IM11" xmlns:mml="http://www.w3.org/1998/Math/MathML" display="inline"><mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mn>1</mml:mn></mml:msub></mml:mrow></mml:mrow></mml:math></inline-formula> scores are given, using the average across all predictors for the tagging predictions.</p>
        <table-wrap id="btae021-T4">
          <label>Table 4.</label>
          <caption>
            <p>
              <inline-formula id="IE12">
                <mml:math xmlns:mml="http://www.w3.org/1998/Math/MathML" id="IM12" display="inline">
                  <mml:mrow>
                    <mml:mrow>
                      <mml:msub>
                        <mml:mrow>
                          <mml:mi>F</mml:mi>
                        </mml:mrow>
                        <mml:mn>1</mml:mn>
                      </mml:msub>
                    </mml:mrow>
                  </mml:mrow>
                </mml:math>
              </inline-formula> score for different architectures.<xref ref-type="table-fn" rid="tblfn2"><sup>a</sup></xref></p>
          </caption>
          <table frame="hsides" rules="groups">
            <colgroup>
              <col valign="top" align="left" />
              <col valign="top" align="char" char="." />
              <col valign="top" align="char" char="." />
              <col valign="top" align="char" char="." />
              <col valign="top" align="char" char="." />
            </colgroup>
            <thead>
              <tr>
                <th rowspan="2">Architecture</th>
                <th colspan="2" align="center">Tagging average<hr /></th>
                <th colspan="2" align="center">Linking<hr /></th>
              </tr>
              <tr>
                <th>Validation</th>
                <th>Testing</th>
                <th>Validation</th>
                <th>Testing</th>
              </tr>
            </thead>
            <tbody>
              <tr>
                <td>BioBERT</td>
                <td>0.764</td>
                <td>0.781</td>
                <td>0.902</td>
                <td>0.944</td>
              </tr>
              <tr>
                <td>BioBERT Split</td>
                <td>0.763</td>
                <td>0.792</td>
                <td>0.877</td>
                <td>0.939</td>
              </tr>
              <tr>
                <td>BioClinicalBERT</td>
                <td>0.734</td>
                <td>0.751</td>
                <td>0.797</td>
                <td>0.908</td>
              </tr>
              <tr>
                <td>BERTBase</td>
                <td>0.707</td>
                <td>0.728</td>
                <td>0.853</td>
                <td>0.885</td>
              </tr>
            </tbody>
          </table>
          <table-wrap-foot>
            <fn id="tblfn2">
              <label>a</label>
              <p>Previously mentioned results are from the ‘BioBERT’-based model. ‘BioBERT Split’ denotes the variant where each tagging predictor has its own pre-trained transformer fine-tuned with it, and the link predictor is trained without the additional same-tag auxiliary classifier.</p>
            </fn>
          </table-wrap-foot>
        </table-wrap>
      </sec>
      <sec>
        <title>4.2 Extrinsic evaluation</title>
        <p>To validate our methodology extrinsically, we carried out a case study using abstracts derived from a broad spectrum of 20 cell biology, genetics, cancer, and multidisciplinary journals. These were ‘Autophagy’, ‘Cancer Research’, ‘Genes and Development’, ‘Cell Death and Disease’, ‘Cell Death and Differentiation’, ‘Apoptosis’, ‘Cell’, ‘Cancer Cell’, ‘Nature’, ‘Nature Cell Biology’, ‘Nature Genetics’, ‘Nature Medicine’, ‘Nature Cancer’, ‘Science’, ‘Science Advances’, ‘eLife’, ‘Journal of Cell Biology’, ‘Journal of Cell Science’, ‘Cell Stem Cell’, and ‘Molecular Cell’. Using the PubMed Advanced Search tool, we selected abstracts that included any one of the selected journal titles as a journal name but excluded ‘review’ as a publication type. The first 10 000 abstracts listed were downloaded as a single text file (including metadata) and were used to test the performance of the model. Of these abstracts, 1526 contained one or more recognizable gene perturbations, with a mean of 1.91 (SD 1.21) gene perturbations per abstract.</p>
        <p>In total, we extracted [we used BERN2 (<xref ref-type="bibr" rid="btae021-B34">Sung <italic>et al.</italic> 2022</xref>) to extract the mentions of genes] 2919 examples of genes that were each associated with a perturbing action. Many genes associated with perturbing actions were also linked to at least one phenotype and/or at least one context. Two hundred and thirty-one genes associated with a perturbing action were linked to a context, an effect and a phenotype, i.e. an example of every entity group, providing the complete set of information required to describe the function of a gene in a specific context.</p>
        <sec>
          <title>4.2.1 Precision of biological entity recognition and labelling</title>
          <p>We selected the entities derived from the group of 231 case study genes that were associated with both a perturbing action and at least one context, one effect, and one phenotype. Some genes were associated with multiple entities in the Context, Effect, and/or Phenotype groups. To calculate the precision of entity recognition, correctly recognized entities were manually scored as TP and entities that should not have been extracted were scored as FP. Perturbing actions, Genes, Contexts, Effects, and Phenotypes were scored as separate groups. Values for precision for all groups were &gt;0.9 (<xref ref-type="table" rid="btae021-T5">Table 5</xref>). Correctly recognized entities were then scored for entity labelling, where entities labelled with the correct assertion(s) were scored as TP and entities labelled with one or more incorrect assertions were labelled as FP. These data were used to calculate scores for the precision of entity labelling, generating values of &gt;0.9 for every group (<xref ref-type="table" rid="btae021-T5">Table 5</xref>). These results show that our model is capable of a high degree of precision in both the recognition and correct labelling of relevant biomedical entities (<xref ref-type="table" rid="btae021-T5">Table 5</xref>). Of note, in the Contexts group, the model achieved high precision values for both entity recognition (0.99) and labelling (0.93), despite the diverse and often idiosyncratic vocabulary used to describe different body tissues, cell lines, and animal strains.</p>
          <table-wrap id="btae021-T5">
            <label>Table 5.</label>
            <caption>
              <p>Precision of entity recognition tagging and entity labelling.<xref ref-type="table-fn" rid="tblfn3"><sup>a</sup></xref></p>
            </caption>
            <table frame="hsides" rules="groups">
              <colgroup>
                <col valign="top" align="left" />
                <col valign="top" align="center" />
                <col valign="top" align="center" />
                <col valign="top" align="center" />
                <col valign="top" align="center" />
                <col valign="top" align="center" />
                <col valign="top" align="center" />
              </colgroup>
              <thead>
                <tr>
                  <th colspan="7">A: Precision of entity recognition tagging<hr /></th>
                </tr>
                <tr>
                  <th></th>
                  <th>PA</th>
                  <th>
                    <italic>G</italic>
                  </th>
                  <th>
                    <italic>E</italic>
                  </th>
                  <th>Ph</th>
                  <th>
                    <italic>C</italic>
                  </th>
                  <th>All</th>
                </tr>
              </thead>
              <tbody>
                <tr>
                  <td>TP</td>
                  <td>214</td>
                  <td>231</td>
                  <td>230</td>
                  <td>214</td>
                  <td>229</td>
                  <td>1118</td>
                </tr>
                <tr>
                  <td>FP</td>
                  <td>17</td>
                  <td>0</td>
                  <td>1</td>
                  <td>17</td>
                  <td>2</td>
                  <td>37</td>
                </tr>
                <tr>
                  <td>Precision</td>
                  <td>
                    <bold>0.93</bold>
                  </td>
                  <td>
                    <bold>1.00</bold>
                  </td>
                  <td>
                    <bold>1.00</bold>
                  </td>
                  <td>
                    <bold>0.93</bold>
                  </td>
                  <td>
                    <bold>0.99</bold>
                  </td>
                  <td>
                    <bold>0.97</bold>
                  </td>
                </tr>
              </tbody>
            </table>
            <table frame="hsides" rules="groups">
              <colgroup>
                <col valign="top" align="left" />
                <col valign="top" align="center" />
                <col valign="top" align="center" />
                <col valign="top" align="center" />
                <col valign="top" align="center" />
                <col valign="top" align="center" />
                <col valign="top" align="center" />
              </colgroup>
              <thead>
                <tr>
                  <th colspan="7">B: Precision of entity labelling<hr /></th>
                </tr>
                <tr>
                  <th></th>
                  <th>PA</th>
                  <th>
                    <italic>G</italic>
                  </th>
                  <th>
                    <italic>E</italic>
                  </th>
                  <th>Ph</th>
                  <th>
                    <italic>C</italic>
                  </th>
                  <th>All</th>
                </tr>
              </thead>
              <tbody>
                <tr>
                  <td>TP</td>
                  <td>205</td>
                  <td>214</td>
                  <td>223</td>
                  <td>202</td>
                  <td>213</td>
                  <td>1057</td>
                </tr>
                <tr>
                  <td>FP</td>
                  <td>9</td>
                  <td>17</td>
                  <td>7</td>
                  <td>12</td>
                  <td>16</td>
                  <td>61</td>
                </tr>
                <tr>
                  <td>Precision</td>
                  <td>
                    <bold>0.96</bold>
                  </td>
                  <td>
                    <bold>0.93</bold>
                  </td>
                  <td>
                    <bold>0.97</bold>
                  </td>
                  <td>
                    <bold>0.94</bold>
                  </td>
                  <td>
                    <bold>0.93</bold>
                  </td>
                  <td>
                    <bold>0.95</bold>
                  </td>
                </tr>
              </tbody>
            </table>
            <table-wrap-foot>
              <fn id="tblfn3">
                <label>a</label>
                <p>TP, true positive; FP, false positive; PA, perturbing actions; <italic>G</italic>, genes; <italic>E</italic>, effects; Ph, phenotypes; <italic>C</italic>, contexts. Precision (values in bold) refers to the number of true positives divided by the total number of positive predictions. </p>
              </fn>
            </table-wrap-foot>
          </table-wrap>
          <p>We used the same case study dataset to examine the distribution of labelled assertions within the different groups. The Perturbing actions group contained representatives from all six assertion groups (as described in <xref ref-type="table" rid="btae021-T1">Table 1</xref>). ‘Gene loss-of-function’ occurred with the highest frequency (91/231) and ‘Pharmacological augmentation’ occurred with the lowest frequency (6/231).</p>
          <p>The Contexts and Phenotypes groups both contained representatives from most, but not all, assertion groups.</p>
          <p>In the Contexts group, the most frequently occurring assertion was ‘Cells’ (50/231) and the second most frequently occurring was ‘Cells; Organism’ (32/231), i.e. studies which involved at least two experimental contexts. The assertion ‘Transformed cells’ (i.e. cancer cells) was highly represented both alone (19/231) and in combination with other assertions, such as ‘Cells’ (5/231). These results show that we were able to extract entities corresponding to a broad range of experimental contexts. The assertion ‘Patient’ was not represented, either alone or in combination with other assertions. This likely reflects the fact that human gene perturbation experiments are more often carried out in either cell lines or patient-derived cells, which would be represented within either the ‘Cells’ or ‘Transformed cells’ assertion groups.</p>
          <p>Within the Phenotype group, the most frequently represented assertion terms were ‘Apoptosis’ (32/231), ‘Proliferation’ (23/231), and ‘Tumourigenesis’ (20/231). Some assertion terms, including ‘Anoikis’, ‘Ferroptosis’, and ‘Quiescence’ were not represented (though a proportion of these were represented within the larger, unfiltered case study dataset of entities associated with 2919 genes). These results suggest that our model could be further refined by training it on a larger and more diverse corpus of annotated abstracts containing a broader range of phenotype terms.</p>
          <p>In cancer genomics, it is of particular interest to identify those genes, which can induce cell death in cancer cells but which do not induce death of normal, healthy cells. To explore the potential of our methodology to identify such genes, we focused on the cell death phenotype ‘Apoptosis’, which was represented by a high number of entities within our case study data. Within the filtered dataset, there were 52 perturbing action and gene mentions associated with ’Apoptosis’ (<xref ref-type="table" rid="btae021-T6">Table 6</xref>). These genes were connected to a broad range of context assertion terms, consistent with the fact that apoptosis is a common form of cell death that has been observed in a many different experimental models.</p>
          <table-wrap id="btae021-T6">
            <label>Table 6.</label>
            <caption>
              <p>Experimental gene perturbations associated with ‘Apoptosis’.</p>
            </caption>
            <table frame="hsides" rules="groups">
              <colgroup>
                <col valign="top" align="left" />
                <col valign="top" align="char" char="." />
              </colgroup>
              <tbody>
                <tr>
                  <td>Total genes associated with phenotype ‘Apoptosis’</td>
                  <td>52</td>
                </tr>
                <tr>
                  <td>Genes associated with <italic>in vitro</italic> assertions (cells; transformed cells; <italic>in vitro</italic>; organoid)</td>
                  <td>45</td>
                </tr>
                <tr>
                  <td>Genes associated with <italic>in vivo</italic> assertions (organism; neoplasm, tissue/organ; <italic>in vivo</italic>; xenograft)</td>
                  <td>25</td>
                </tr>
                <tr>
                  <td>Overlap (genes associated with <italic>in vitro</italic> assertions that are also associated with <italic>in vivo</italic> assertions)</td>
                  <td>18</td>
                </tr>
                <tr>
                  <td>Genes associated with non-cancer assertions (organism, tissue/organ, cells, organoid, <italic>in vitro</italic>, <italic>in vivo</italic>)</td>
                  <td>41</td>
                </tr>
                <tr>
                  <td>Genes associated with cancer assertions (transformed cells; neoplasm; xenograft)</td>
                  <td>22</td>
                </tr>
                <tr>
                  <td>Overlap (genes associated with non-cancer assertions that are also associated with cancer assertions)</td>
                  <td>10</td>
                </tr>
              </tbody>
            </table>
          </table-wrap>
          <p>The case study sample size was too small to allow in-depth analysis of gene perturbations in individual contexts. To further analyse the results, we therefore grouped the apoptosis-associated genes in two separate ways; (i) genes associated with <italic>in vitro</italic> assertion terms compared with genes associated with <italic>in vivo</italic> assertion terms and (ii) genes associated with non-cancer assertion terms and compared genes associated with cancer assertion terms (as detailed in <xref ref-type="table" rid="btae021-T1">Table 1</xref>). Within each of these two pairs of groups, there were some genes that were present within only one group and others which were present within both groups (<xref ref-type="table" rid="btae021-T6">Table 6</xref>). For example, in the <italic>in vitro</italic> group versus <italic>in vivo</italic> group pairing, the genes BCL-6, FasL, and BAX were associated with apoptosis in <italic>in vitro</italic> contexts only, Tctp, PKD1, and Ldlr were associated with apoptosis in <italic>in vivo</italic> contexts only, and the genes Ang2, Foxo3a, mTOR, Cox-2, and Smad2 were associated with apoptosis in both <italic>in vitro</italic> and <italic>in vivo</italic> contexts. In the non-cancer group versus cancer group pairing, the genes Ang2, CD28, and Foxo3a were associated with apoptosis in non-cancer contexts only, Pim1, Hsp90, and EGFR were associated with apoptosis in cancer contexts only, and p53, RAB25, and E2F-1 were associated with apoptosis in both non-cancer and cancer contexts. Thus, our methodology was able to identify genes involved in the regulation of apoptosis in a range of contexts and, when applied to a larger sample size, could be a useful means to explore variation of regulatory mechanisms in different cell types and experimental contexts.</p>
          <p>Although the small sample size used in our case study precludes drawing any specific biological inferences, our results provide a proof-of-principle demonstration of how our new methodology could be used to generate insights into an important real-world question in cancer genomics.</p>
        </sec>
      </sec>
    </sec>
    <sec>
      <title>5 Discussion</title>
      <p>Genomics research papers are superabundant within the biomedical domain of digital literature and their content is critically important both to the understanding of basic biological processes and to the design of new therapeutic strategies for cancer and other human diseases. There is an unmet need in biomedical TM for tailored methodologies that can systematically and accurately extract the most important details of genomics experiments from unstructured text, present them to researchers in an accessible form and thereby maximize the value of the existing knowledge base.</p>
      <p>We have shown here that our new model can learn to recognize the five different classes of biological entity required to extract the results of genetics experiments from biomedical texts: genes, perturbing actions, effects, phenotypes, and contexts. We can automatically label the extracted entities with assertion terms that further define their meaning. Intrinsic evaluation has shown that our approach achieves a consistent performance of <inline-formula id="IE13"><mml:math id="IM13" xmlns:mml="http://www.w3.org/1998/Math/MathML" display="inline"><mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mn>1</mml:mn></mml:msub></mml:mrow></mml:mrow></mml:math></inline-formula> score between 0.75 and 0.79 across all categories. In addition, our methodology can accurately identify relations between the different groups of biological entities, with an <inline-formula id="IE14"><mml:math id="IM14" xmlns:mml="http://www.w3.org/1998/Math/MathML" display="inline"><mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mn>1</mml:mn></mml:msub></mml:mrow></mml:mrow></mml:math></inline-formula> score of 0.9. The accuracy of our work measures favourably against other comparable corpora, e.g. the Phenotype Gene Relations corpus (<xref ref-type="bibr" rid="btae021-B30">Sousa <italic>et al.</italic> 2019</xref>) and released models, has a maximum <inline-formula id="IE15"><mml:math id="IM15" xmlns:mml="http://www.w3.org/1998/Math/MathML" display="inline"><mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mn>1</mml:mn></mml:msub></mml:mrow></mml:mrow></mml:math></inline-formula> score of 0.68. Similarly, the model by <xref ref-type="bibr" rid="btae021-B37">Xing <italic>et al.</italic> (2018)</xref>, which also extracts phenotype and genotype relationships from text has a best <inline-formula id="IE16"><mml:math id="IM16" xmlns:mml="http://www.w3.org/1998/Math/MathML" display="inline"><mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mn>1</mml:mn></mml:msub></mml:mrow></mml:mrow></mml:math></inline-formula> score of 0.67. In addition to the intrinsic evaluation, we have also demonstrated the accuracy of our approach in a real-world case study achieving a precision score over 0.9. Some key attributes of our model are that it is relatively lightweight and that it is able to run on a large quantity of data with a high level of accuracy.</p>
      <p>The novel ability of our methodology to reliably and accurately recognize methods of gene perturbation and experimental contexts, which are both diverse and continuously evolving, can allow researchers to analyse the impact of genes in different experimental approaches and in different species and cell types. The methodology generalizes to various subdomains including basic cell biology, basic and applied cancer research, and multidisciplinary texts, enabling broad coverage of biomedical literature. It is also potentially adaptable to other related applications, e.g. the investigation of the impacts of genes or drugs on downstream gene expression or disease aetiologies.</p>
      <p>The overall results of our evaluation demonstrate that our approach can accurately extract functional genomics relationships and relevant context from PubMed abstract text. In future work, we propose two key improvements. Firstly, it would be of high value to adapt our machine-learning models to work on full article texts, thereby providing access to a much greater volume of relevant information and context for extraction. Full-text classifier adaptation is possible and has been demonstrated in similar work (<xref ref-type="bibr" rid="btae021-B22">Oliveira Gonçalves <italic>et al.</italic> 2021</xref>, <xref ref-type="bibr" rid="btae021-B12">Gonçalves <italic>et al.</italic> 2022</xref>). Secondly, our methodology would benefit from linking the identified phenotypes to the recently available HPO (<xref ref-type="bibr" rid="btae021-B13">Groza <italic>et al.</italic> 2015</xref>). This will allow extracted phenotypes to be normalized to a standardized vocabulary, as well as enabling the integration of additional contextual information provided by HPO for disease phenotypes.</p>
    </sec>
    <sec>
      <title>6 Conclusion</title>
      <p>In this article, we have presented the first TM methodology that is specifically designed for extracting the contexts of genomics experiments. We have developed a NER model as well as a relation extraction model that can identify the key components of genomics experiments, including contexts, and have evaluated our models both intrinsically and extrinsically. Our approach leverages the latest developments in Natural Language Processing and Machine Learning and outperforms the <inline-formula id="IE17"><mml:math id="IM17" xmlns:mml="http://www.w3.org/1998/Math/MathML" display="inline"><mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mn>1</mml:mn></mml:msub></mml:mrow></mml:mrow></mml:math></inline-formula> scores of earlier models. We believe that our methodology has the potential to significantly enhance and accelerate researchers’ ability to access and contextualize genomics information compared to previously existing TM tools.</p>
    </sec>
  </body>
  <back>
    <sec>
      <title>Supplementary data</title>
      <p>
        <xref ref-type="supplementary-material" rid="sup1">Supplementary data</xref> are available at <italic>Bioinformatics</italic> online.</p>
    </sec>
    <sec>
      <title>Conflict of interest</title>
      <p>None declared.</p>
    </sec>
    <sec>
      <title>Funding</title>
      <p>This work was supported by grant UK Research and Innovation EP/Y031350/1 and an Amazon Machine Learning Research Award to A.K. M.N. is supported by the Cancer Research UK Cambridge Institute core grant C9545/A29580; Biotechnology and Biological Sciences Research Council grants BB/S013466/1, BB/T013486/1; and Diabetes UK via BIRAX and the British Council [65BX18MNIB].</p>
    </sec>
    <ref-list id="ref1">
      <title>References</title>
      <ref id="btae021-B1">
        <mixed-citation publication-type="journal">
          <person-group person-group-type="author">
            <string-name name-style="western">
              <surname>Ananiadou</surname>
              <given-names>S</given-names>
            </string-name>, <string-name name-style="western"><surname>Thompson</surname><given-names>P</given-names></string-name>, <string-name name-style="western"><surname>Nawaz</surname><given-names>R</given-names></string-name></person-group>
          <etal>et al</etal>
          <article-title>Event-based text mining for biology and functional genomics</article-title>. <source>Brief Funct Genomics</source><year>2015</year>;<volume>14</volume>:<fpage>213</fpage>–<lpage>30</lpage>.</mixed-citation>
      </ref>
      <ref id="btae021-B2">
        <mixed-citation publication-type="journal">
          <person-group person-group-type="author">
            <string-name name-style="western">
              <surname>Ashburner</surname>
              <given-names>M</given-names>
            </string-name>, <string-name name-style="western"><surname>Ball</surname><given-names>CA</given-names></string-name>, <string-name name-style="western"><surname>Blake</surname><given-names>JA</given-names></string-name></person-group>
          <etal>et al</etal>
          <article-title>Gene ontology: tool for the unification of biology</article-title>. <source>Nat Genet</source><year>2000</year>;<volume>25</volume>:<fpage>25</fpage>–<lpage>9</lpage>.</mixed-citation>
      </ref>
      <ref id="btae021-B3">
        <mixed-citation publication-type="journal">
          <person-group person-group-type="author">
            <string-name name-style="western">
              <surname>Baker</surname>
              <given-names>S</given-names>
            </string-name>, <string-name name-style="western"><surname>Ali</surname><given-names>I</given-names></string-name>, <string-name name-style="western"><surname>Silins</surname><given-names>I</given-names></string-name></person-group>
          <etal>et al</etal>
          <article-title>Cancer Hallmarks Analytics Tool (CHAT): a text mining approach to organize and evaluate scientific literature on cancer</article-title>. <source>Bioinformatics</source><year>2017</year>;<volume>33</volume>:<fpage>3973</fpage>–<lpage>81</lpage>.</mixed-citation>
      </ref>
      <ref id="btae021-B4">
        <mixed-citation publication-type="journal">
          <person-group person-group-type="author">
            <string-name name-style="western">
              <surname>Baker</surname>
              <given-names>S</given-names>
            </string-name>, <string-name name-style="western"><surname>Silins</surname><given-names>I</given-names></string-name>, <string-name name-style="western"><surname>Guo</surname><given-names>Y</given-names></string-name></person-group>
          <etal>et al</etal>
          <article-title>Automatic semantic classification of scientific literature according to the hallmarks of cancer</article-title>. <source>Bioinformatics</source><year>2016</year>;<volume>32</volume>:<fpage>432</fpage>–<lpage>40</lpage>.</mixed-citation>
      </ref>
      <ref id="btae021-B5">
        <mixed-citation publication-type="journal">
          <person-group person-group-type="author">
            <string-name name-style="western">
              <surname>Berger</surname>
              <given-names>MF</given-names>
            </string-name>, <string-name name-style="western"><surname>Mardis</surname><given-names>ER.</given-names></string-name></person-group>
          <article-title>The emerging clinical relevance of genomics in cancer medicine</article-title>. <source>Nat Rev Clin Oncol</source><year>2018</year>;<volume>15</volume>:<fpage>353</fpage>–<lpage>65</lpage>.</mixed-citation>
      </ref>
      <ref id="btae021-B6">
        <mixed-citation publication-type="book">
          <person-group person-group-type="author">
            <string-name name-style="western">
              <surname>Celis</surname>
              <given-names>UM</given-names>
            </string-name>, <string-name name-style="western"><surname>García-Gasca</surname><given-names>T</given-names></string-name>, <string-name name-style="western"><surname>Mejía</surname><given-names>C.</given-names></string-name></person-group>
          <source>Apoptosis-induced Compensatory Proliferation in Cancer</source>. In: Sergi CM (Ed.) <italic>Metastasis</italic>. <publisher-name>Exon Publications</publisher-name>, Brisbane, Australia <year>2022</year>, <fpage>149</fpage>–<lpage>61</lpage>.</mixed-citation>
      </ref>
      <ref id="btae021-B7">
        <mixed-citation publication-type="journal">
          <person-group person-group-type="author">
            <string-name name-style="western">
              <surname>Chun</surname>
              <given-names>H-W</given-names>
            </string-name>, <string-name name-style="western"><surname>Tsuruoka</surname><given-names>Y</given-names></string-name>, <string-name name-style="western"><surname>Kim</surname><given-names>J-D</given-names></string-name></person-group>
          <etal>et al</etal>
          <article-title>Extraction of gene-disease relations from Medline using domain dictionaries and machine learning</article-title>. <source> In: </source><italic>Pac Symp Biocomput Maui, Hawaii USA, 3–7 January</italic> <year>2006</year>;<fpage>4</fpage>–<lpage>15</lpage>.</mixed-citation>
      </ref>
      <ref id="btae021-B8">
        <mixed-citation publication-type="journal">
          <person-group person-group-type="author">
            <string-name name-style="western">
              <surname>Demaine</surname>
              <given-names>J</given-names>
            </string-name>, <string-name name-style="western"><surname>Martin</surname><given-names>J</given-names></string-name>, <string-name name-style="western"><surname>Wei</surname><given-names>L</given-names></string-name></person-group>
          <etal>et al</etal>
          <article-title>LitMiner: integration of library services within a bio-informatics application</article-title>. <source>Biomed Digit Libr</source><year>2006</year>;<volume>3</volume>:<fpage>11</fpage>–<lpage>8</lpage>.</mixed-citation>
      </ref>
      <ref id="btae021-B9">
        <mixed-citation publication-type="journal">
          <person-group person-group-type="author">
            <string-name name-style="western">
              <surname>Demaria</surname>
              <given-names>M</given-names>
            </string-name>, <string-name name-style="western"><surname>O'Leary</surname><given-names>MN</given-names></string-name>, <string-name name-style="western"><surname>Chang</surname><given-names>J</given-names></string-name></person-group>
          <etal>et al</etal>
          <article-title>Cellular senescence promotes adverse effects of chemotherapy and cancer relapsecellular senescence and chemotherapy</article-title>. <source>Cancer Discov</source><year>2017</year>;<volume>7</volume>:<fpage>165</fpage>–<lpage>76</lpage>.</mixed-citation>
      </ref>
      <ref id="btae021-B10">
        <mixed-citation publication-type="other">
          <person-group person-group-type="author">
            <string-name name-style="western">
              <surname>Devlin</surname>
              <given-names>J</given-names>
            </string-name>, <string-name name-style="western"><surname>Chang</surname><given-names>M-W</given-names></string-name>, <string-name name-style="western"><surname>Lee</surname><given-names>K</given-names></string-name></person-group>
          <etal>et al</etal> BERT: pre-training of deep bidirectional transformers for language understanding. arXiv:1810.04805v2, <year>2019</year>.</mixed-citation>
      </ref>
      <ref id="btae021-B11">
        <mixed-citation publication-type="journal">
          <person-group person-group-type="author">
            <string-name name-style="western">
              <surname>Galluzzi</surname>
              <given-names>L</given-names>
            </string-name>, <string-name name-style="western"><surname>Vitale</surname><given-names>I</given-names></string-name>, <string-name name-style="western"><surname>Aaronson</surname><given-names>SA</given-names></string-name></person-group>
          <etal>et al</etal>
          <article-title>Molecular mechanisms of cell death: recommendations of the nomenclature committee on cell death 2018</article-title>. <source>Cell Death Differ</source><year>2018</year>;<volume>25</volume>:<fpage>486</fpage>–<lpage>541</lpage>.</mixed-citation>
      </ref>
      <ref id="btae021-B12">
        <mixed-citation publication-type="journal">
          <person-group person-group-type="author">
            <string-name name-style="western">
              <surname>Gonçalves</surname>
              <given-names>CA</given-names>
            </string-name>, <string-name name-style="western"><surname>Vieira</surname><given-names>AS</given-names></string-name>, <string-name name-style="western"><surname>Gonçalves</surname><given-names>CT</given-names></string-name></person-group>
          <etal>et al</etal>
          <article-title>A novel multi-view ensemble learning architecture to improve the structured text classification</article-title>. <source>Information</source><year>2022</year>;<volume>13</volume>:<fpage>283</fpage>.</mixed-citation>
      </ref>
      <ref id="btae021-B13">
        <mixed-citation publication-type="journal">
          <person-group person-group-type="author">
            <string-name name-style="western">
              <surname>Groza</surname>
              <given-names>T</given-names>
            </string-name>, <string-name name-style="western"><surname>Köhler</surname><given-names>S</given-names></string-name>, <string-name name-style="western"><surname>Moldenhauer</surname><given-names>D</given-names></string-name></person-group>
          <etal>et al</etal>
          <article-title>The human phenotype ontology: semantic unification of common and rare disease</article-title>. <source>Am J Hum Genet</source><year>2015</year>;<volume>97</volume>:<fpage>111</fpage>–<lpage>24</lpage>.</mixed-citation>
      </ref>
      <ref id="btae021-B14">
        <mixed-citation publication-type="journal">
          <person-group person-group-type="author">
            <string-name name-style="western">
              <surname>Kerr</surname>
              <given-names>JF</given-names>
            </string-name>, <string-name name-style="western"><surname>Wyllie</surname><given-names>AH</given-names></string-name>, <string-name name-style="western"><surname>Currie</surname><given-names>AR.</given-names></string-name></person-group>
          <article-title>Apoptosis: a basic biological phenomenon with wideranging implications in tissue kinetics</article-title>. <source>Br J Cancer</source><year>1972</year>;<volume>26</volume>:<fpage>239</fpage>–<lpage>57</lpage>.</mixed-citation>
      </ref>
      <ref id="btae021-B15">
        <mixed-citation publication-type="journal">
          <person-group person-group-type="author">
            <string-name name-style="western">
              <surname>Kuleshov</surname>
              <given-names>MV</given-names>
            </string-name>, <string-name name-style="western"><surname>Jones</surname><given-names>MR</given-names></string-name>, <string-name name-style="western"><surname>Rouillard</surname><given-names>AD</given-names></string-name></person-group>
          <etal>et al</etal>
          <article-title>Enrichr: a comprehensive gene set enrichment analysis web server 2016 update</article-title>. <source>Nucleic Acids Res</source><year>2016</year>;<volume>44</volume>:<fpage>W90</fpage>–<lpage>7</lpage>.</mixed-citation>
      </ref>
      <ref id="btae021-B16">
        <mixed-citation publication-type="journal">
          <person-group person-group-type="author">
            <string-name name-style="western">
              <surname>Labi</surname>
              <given-names>V</given-names>
            </string-name>, <string-name name-style="western"><surname>Erlacher</surname><given-names>M.</given-names></string-name></person-group>
          <article-title>How cell death shapes cancer</article-title>. <source>Cell Death Dis</source><year>2015</year>;<volume>6</volume>:<fpage>e1675</fpage>.</mixed-citation>
      </ref>
      <ref id="btae021-B17">
        <mixed-citation publication-type="journal">
          <person-group person-group-type="author">
            <string-name name-style="western">
              <surname>Lander</surname>
              <given-names>ES</given-names>
            </string-name>, <string-name name-style="western"><surname>Linton</surname><given-names>LM</given-names></string-name>, <string-name name-style="western"><surname>Birren</surname><given-names>B</given-names></string-name></person-group>
          <etal>et al</etal>; <collab>International Human Genome Sequencing Consortium</collab>. <article-title>Initial sequencing and analysis of the human genome</article-title>. <source>Nature</source><year>2001</year>;<volume>409</volume>:<fpage>860</fpage>–<lpage>921</lpage>.</mixed-citation>
      </ref>
      <ref id="btae021-B18">
        <mixed-citation publication-type="journal">
          <person-group person-group-type="author">
            <string-name name-style="western">
              <surname>Lee</surname>
              <given-names>J</given-names>
            </string-name>, <string-name name-style="western"><surname>Yoon</surname><given-names>W</given-names></string-name>, <string-name name-style="western"><surname>Kim</surname><given-names>S</given-names></string-name></person-group>
          <etal>et al</etal>
          <article-title>BioBERT: a pre-trained biomedical language representation model for biomedical text mining</article-title>. <source>Bioinformatics</source><year>2020</year>;<volume>36</volume>:<fpage>1234</fpage>–<lpage>40</lpage>.</mixed-citation>
      </ref>
      <ref id="btae021-B19">
        <mixed-citation publication-type="journal">
          <person-group person-group-type="author">
            <string-name name-style="western">
              <surname>Luo</surname>
              <given-names>H</given-names>
            </string-name>, <string-name name-style="western"><surname>Lin</surname><given-names>Y</given-names></string-name>, <string-name name-style="western"><surname>Liu</surname><given-names>T</given-names></string-name></person-group>
          <etal>et al</etal>
          <article-title>DEG 15, an update of the Database of Essential Genes that includes built-in analysis tools</article-title>. <source>Nucleic Acids Res</source><year>2021</year>;<volume>49</volume>:<fpage>D677</fpage>–<lpage>86</lpage>.</mixed-citation>
      </ref>
      <ref id="btae021-B20">
        <mixed-citation publication-type="journal">
          <person-group person-group-type="author">
            <string-name name-style="western">
              <surname>Miller</surname>
              <given-names>DM</given-names>
            </string-name>, <string-name name-style="western"><surname>Shalhout</surname><given-names>SZ.</given-names></string-name></person-group>
          <article-title>GENETEX—a genomics report text mining r package and shiny application designed to capture real-world clinico-genomic data</article-title>. <source>JAMIA Open</source><year>2021</year>;<volume>4</volume>:<fpage>ooab082</fpage>.</mixed-citation>
      </ref>
      <ref id="btae021-B21">
        <mixed-citation publication-type="journal">
          <person-group person-group-type="author">
            <string-name name-style="western">
              <surname>Ohsumi</surname>
              <given-names>Y.</given-names>
            </string-name>
          </person-group>
          <article-title>Historical landmarks of autophagy research</article-title>. <source>Cell Res</source><year>2014</year>;<volume>24</volume>:<fpage>9</fpage>–<lpage>23</lpage>.</mixed-citation>
      </ref>
      <ref id="btae021-B22">
        <mixed-citation publication-type="journal">
          <person-group person-group-type="author">
            <string-name name-style="western">
              <surname>Oliveira Gonçalves</surname>
              <given-names>CA</given-names>
            </string-name>, <string-name name-style="western"><surname>Camacho</surname><given-names>R</given-names></string-name>, <string-name name-style="western"><surname>Gonçalves</surname><given-names>CT</given-names></string-name></person-group>
          <etal>et al</etal>
          <article-title>Classification of full text biomedical documents: sections importance assessment</article-title>. <source>Appl Sci</source><year>2021</year>;<volume>11</volume>:<fpage>2674</fpage>.</mixed-citation>
      </ref>
      <ref id="btae021-B23">
        <mixed-citation publication-type="other">
          <person-group person-group-type="author">
            <string-name name-style="western">
              <surname>Pafilis</surname>
              <given-names>E</given-names>
            </string-name>, <string-name name-style="western"><surname>Pavlopoulos</surname><given-names>GA</given-names></string-name>, <string-name name-style="western"><surname>Satagopam</surname><given-names>VP</given-names></string-name></person-group>
          <etal>et al</etal> OnTheFly 2.0: a tool for automatic annotation of files and biological information extraction. In: <italic>13th IEEE International Conference on BioInformatics and BioEngineering Chania, Greece Nov 10th-13th 2013</italic>. 1-4. IEEE Computer Society 2013.</mixed-citation>
      </ref>
      <ref id="btae021-B24">
        <mixed-citation publication-type="journal">
          <person-group person-group-type="author">
            <string-name name-style="western">
              <surname>Perfetto</surname>
              <given-names>L</given-names>
            </string-name>, <string-name name-style="western"><surname>Briganti</surname><given-names>L</given-names></string-name>, <string-name name-style="western"><surname>Calderone</surname><given-names>A</given-names></string-name></person-group>
          <etal>et al</etal>
          <article-title>SIGNOR: a database of causal relationships between biological entities</article-title>. <source>Nucleic Acids Res</source><year>2016</year>;<volume>44</volume>:<fpage>D548</fpage>–<lpage>54</lpage>.</mixed-citation>
      </ref>
      <ref id="btae021-B25">
        <mixed-citation publication-type="journal">
          <person-group person-group-type="author">
            <string-name name-style="western">
              <surname>Przybyla</surname>
              <given-names>L</given-names>
            </string-name>, <string-name name-style="western"><surname>Gilbert</surname><given-names>LA.</given-names></string-name></person-group>
          <article-title>A new era in functional genomics screens</article-title>. <source>Nat Rev Genet</source><year>2022</year>;<volume>23</volume>:<fpage>89</fpage>–<lpage>103</lpage>.</mixed-citation>
      </ref>
      <ref id="btae021-B26">
        <mixed-citation publication-type="journal">
          <person-group person-group-type="author">
            <string-name name-style="western">
              <surname>Pyysalo</surname>
              <given-names>S</given-names>
            </string-name>, <string-name name-style="western"><surname>Baker</surname><given-names>S</given-names></string-name>, <string-name name-style="western"><surname>Ali</surname><given-names>I</given-names></string-name></person-group>
          <etal>et al</etal>
          <article-title>LION LBD: a literature-based discovery system for cancer biology</article-title>. <source>Bioinformatics</source><year>2019</year>;<volume>35</volume>:<fpage>1553</fpage>–<lpage>61</lpage>.</mixed-citation>
      </ref>
      <ref id="btae021-B27">
        <mixed-citation publication-type="journal">
          <person-group person-group-type="author">
            <string-name name-style="western">
              <surname>Renganathan</surname>
              <given-names>V.</given-names>
            </string-name>
          </person-group>
          <article-title>Text mining in biomedical domain with emphasis on document clustering</article-title>. <source>Healthc Inform Res</source><year>2017</year>;<volume>23</volume>:<fpage>141</fpage>–<lpage>6</lpage>.</mixed-citation>
      </ref>
      <ref id="btae021-B28">
        <mixed-citation publication-type="other">
          <person-group person-group-type="author">
            <string-name name-style="western">
              <surname>Rim</surname>
              <given-names>K.</given-names>
            </string-name>
          </person-group> MAE2: portable annotation tool for general natural language use. In: <italic>Proceedings 12th Joint ACL-ISO Workshop on Interoperable Semantic Annotation</italic>, <source><italic>28 May 2016,</italic></source><italic>Portoroz, Slovenia</italic><fpage>75</fpage>–<lpage>80</lpage>. Association for Computational Linguistics <year>2016</year>.</mixed-citation>
      </ref>
      <ref id="btae021-B29">
        <mixed-citation publication-type="journal">
          <person-group person-group-type="author">
            <string-name name-style="western">
              <surname>Shimada</surname>
              <given-names>K</given-names>
            </string-name>, <string-name name-style="western"><surname>Bachman</surname><given-names>JA</given-names></string-name>, <string-name name-style="western"><surname>Muhlich</surname><given-names>JL</given-names></string-name></person-group>
          <etal>et al</etal>
          <article-title>shinyDepMap, a tool to identify targetable cancer genes and their functional connections from Cancer Dependency Map data</article-title>. <source>Elife</source><year>2021</year>;<volume>10</volume>:<fpage>e57116</fpage>.</mixed-citation>
      </ref>
      <ref id="btae021-B30">
        <mixed-citation publication-type="other">
          <person-group person-group-type="author">
            <string-name name-style="western">
              <surname>Sousa</surname>
              <given-names>D</given-names>
            </string-name>, <string-name name-style="western"><surname>Lamurias</surname><given-names>A</given-names></string-name>, <string-name name-style="western"><surname>Couto</surname><given-names>FM.</given-names></string-name></person-group> A silver standard corpus of human phenotype-gene relations. In: <italic>Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies</italic>, <italic>Minneapolis, Minnesota USA, June 2nd-7th, 2019</italic>. Vol. <volume>1</volume>. <fpage>1487</fpage>–<lpage>92</lpage>. Association for Computational Linguistics, <year>2019</year>.</mixed-citation>
      </ref>
      <ref id="btae021-B31">
        <mixed-citation publication-type="journal">
          <person-group person-group-type="author">
            <string-name name-style="western">
              <surname>Strasser</surname>
              <given-names>A</given-names>
            </string-name>, <string-name name-style="western"><surname>Vaux</surname><given-names>DL.</given-names></string-name></person-group>
          <article-title>Cell death in the origin and treatment of cancer</article-title>. <source>Mol Cell</source><year>2020</year>;<volume>78</volume>:<fpage>1045</fpage>–<lpage>54</lpage>.</mixed-citation>
      </ref>
      <ref id="btae021-B32">
        <mixed-citation publication-type="other">
          <person-group person-group-type="author">
            <string-name name-style="western">
              <surname>Stubbs</surname>
              <given-names>A.</given-names>
            </string-name>
          </person-group> MAE and MAI: lightweight annotation and adjudication tools. In: <italic>Proceedings of the 5th Linguistic Annotation Workshop 23rd-24th June 2011, Portland, Oregon, USA. 129-33</italic>. Association for Computational Linguistics. <year>2011</year>.</mixed-citation>
      </ref>
      <ref id="btae021-B33">
        <mixed-citation publication-type="journal">
          <person-group person-group-type="author">
            <string-name name-style="western">
              <surname>Sung</surname>
              <given-names>H</given-names>
            </string-name>, <string-name name-style="western"><surname>Ferlay</surname><given-names>J</given-names></string-name>, <string-name name-style="western"><surname>Siegel</surname><given-names>RL</given-names></string-name></person-group>
          <etal>et al</etal>
          <article-title>Global Cancer Statistics 2020: GLOBOCAN estimates of incidence and mortality worldwide for 36 cancers in 185 countries</article-title>. <source>CA Cancer J Clin</source><year>2021</year>;<volume>71</volume>:<fpage>209</fpage>–<lpage>49</lpage>.</mixed-citation>
      </ref>
      <ref id="btae021-B34">
        <mixed-citation publication-type="journal">
          <person-group person-group-type="author">
            <string-name name-style="western">
              <surname>Sung</surname>
              <given-names>M</given-names>
            </string-name>, <string-name name-style="western"><surname>Jeong</surname><given-names>M</given-names></string-name>, <string-name name-style="western"><surname>Choi</surname><given-names>Y</given-names></string-name></person-group>
          <etal>et al</etal>
          <article-title>BERN2: an advanced neural biomedical named entity recognition and normalization tool</article-title>. <source>Bioinformatics</source><year>2022</year>;<volume>38</volume>:<fpage>4837</fpage>–<lpage>9</lpage>.</mixed-citation>
      </ref>
      <ref id="btae021-B35">
        <mixed-citation publication-type="journal">
          <person-group person-group-type="author">
            <string-name name-style="western">
              <surname>Venter</surname>
              <given-names>JC</given-names>
            </string-name>, <string-name name-style="western"><surname>Adams</surname><given-names>MD</given-names></string-name>, <string-name name-style="western"><surname>Myers</surname><given-names>EW</given-names></string-name></person-group>
          <etal>et al</etal>
          <article-title>The sequence of the human genome</article-title>. <source>Science</source><year>2001</year>;<volume>291</volume>:<fpage>1304</fpage>–<lpage>51</lpage>.</mixed-citation>
      </ref>
      <ref id="btae021-B36">
        <mixed-citation publication-type="journal">
          <person-group person-group-type="author">
            <string-name name-style="western">
              <surname>Warde-Farley</surname>
              <given-names>D</given-names>
            </string-name>, <string-name name-style="western"><surname>Donaldson</surname><given-names>SL</given-names></string-name>, <string-name name-style="western"><surname>Comes</surname><given-names>O</given-names></string-name></person-group>
          <etal>et al</etal>
          <article-title>The GeneMANIA prediction server: biological network integration for gene prioritization and predicting gene function</article-title>. <source>Nucleic Acids Res</source><year>2010</year>;<volume>38</volume>:<fpage>W214</fpage>–<lpage>20</lpage>.</mixed-citation>
      </ref>
      <ref id="btae021-B37">
        <mixed-citation publication-type="journal">
          <person-group person-group-type="author">
            <string-name name-style="western">
              <surname>Xing</surname>
              <given-names>W</given-names>
            </string-name>, <string-name name-style="western"><surname>Qi</surname><given-names>J</given-names></string-name>, <string-name name-style="western"><surname>Yuan</surname><given-names>X</given-names></string-name></person-group>
          <etal>et al</etal>
          <article-title>A gene–phenotype relationship extraction pipeline from the biomedical literature using a representation learning approach</article-title>. <source>Bioinformatics</source><year>2018</year>;<volume>34</volume>:<fpage>i386</fpage>–<lpage>94</lpage>.</mixed-citation>
      </ref>
    </ref-list>
  </back>
</article>