<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD v1.0 20120330//EN" "JATS-archivearticle1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink">
  <front>
    <journal-meta />
    <article-meta>
      <title-group>
        <article-title>Application of the RDF framework to integrate heterogenous experimental data of a large chemo- and biodiverse collection from a collaborative research project</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="author">
          <string-name>Frederic Burdet</string-name>
          <xref ref-type="aff" rid="aff8">8</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Luis-Manuel Quiros-Guerrero</string-name>
          <xref ref-type="aff" rid="aff5">5</xref>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Olivier Kirchhofer</string-name>
          <xref ref-type="aff" rid="aff5">5</xref>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Jahn Nitschke</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Pierre-Marie Allard</string-name>
          <xref ref-type="aff" rid="aff1">1</xref>
          <xref ref-type="aff" rid="aff5">5</xref>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Louis-Felix Nothias</string-name>
          <xref ref-type="aff" rid="aff5">5</xref>
          <xref ref-type="aff" rid="aff6">6</xref>
          <xref ref-type="aff" rid="aff7">7</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Arnaud Gaudry</string-name>
          <xref ref-type="aff" rid="aff5">5</xref>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Sébastien Moretti</string-name>
          <xref ref-type="aff" rid="aff8">8</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Robin Engler</string-name>
          <xref ref-type="aff" rid="aff8">8</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Emerson Ferreira Queiroz</string-name>
          <xref ref-type="aff" rid="aff5">5</xref>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Nabil Hanna</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Chunyan Wu</string-name>
          <xref ref-type="aff" rid="aff3">3</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Antonio Grondin</string-name>
          <xref ref-type="aff" rid="aff4">4</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Bruno David</string-name>
          <xref ref-type="aff" rid="aff4">4</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Thierry Soldati</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Christian Wolfrum</string-name>
          <xref ref-type="aff" rid="aff3">3</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Erick Carreira</string-name>
          <xref ref-type="aff" rid="aff2">2</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Jean-Luc Wolfender</string-name>
          <xref ref-type="aff" rid="aff5">5</xref>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Marco Pagni</string-name>
          <xref ref-type="aff" rid="aff8">8</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Florence Mehl</string-name>
          <xref ref-type="aff" rid="aff8">8</xref>
        </contrib>
        <aff id="aff0">
          <label>0</label>
          <institution>Department of Biochemistry, Faculty of Science, University of Geneva</institution>
          ,
          <addr-line>1211 Geneva</addr-line>
          ,
          <country country="CH">Switzerland</country>
        </aff>
        <aff id="aff1">
          <label>1</label>
          <institution>Department of Biology, University of Fribourg</institution>
          ,
          <addr-line>1700 Fribourg</addr-line>
          ,
          <country country="CH">Switzerland</country>
        </aff>
        <aff id="aff2">
          <label>2</label>
          <institution>Department of Chemistry and Applied Biosciences</institution>
          ,
          <addr-line>ETHZ, 8093 Zürich</addr-line>
          ,
          <country country="CH">Switzerland</country>
        </aff>
        <aff id="aff3">
          <label>3</label>
          <institution>Department of Health Sciences and Technology</institution>
          ,
          <addr-line>ETHZ, 8092 Zürich</addr-line>
          ,
          <country country="CH">Switzerland</country>
        </aff>
        <aff id="aff4">
          <label>4</label>
          <institution>Green Mission Pierre Fabre, Institut de Recherche Pierre Fabre</institution>
          ,
          <addr-line>31562 Toulouse</addr-line>
          ,
          <country country="FR">France</country>
        </aff>
        <aff id="aff5">
          <label>5</label>
          <institution>Institute of Pharmaceutical Sciences of Western Switzerland, University of Geneva</institution>
          ,
          <addr-line>1211 Geneva 4</addr-line>
          ,
          <country country="CH">Switzerland</country>
        </aff>
        <aff id="aff6">
          <label>6</label>
          <institution>School of Pharmaceutical Sciences, University of Geneva</institution>
          ,
          <addr-line>1211 Geneva 4</addr-line>
          ,
          <country country="CH">Switzerland</country>
        </aff>
        <aff id="aff7">
          <label>7</label>
          <institution>Université Côte d'Azur, Institut de Chimie de Nice</institution>
          ,
          <addr-line>Campus Valrose, Nice</addr-line>
          ,
          <country country="FR">France</country>
        </aff>
        <aff id="aff8">
          <label>8</label>
          <institution>Vital-IT, SIB Swiss Institute of Bioinformatics</institution>
          ,
          <addr-line>1015 Lausanne</addr-line>
          ,
          <country country="CH">Switzerland</country>
        </aff>
      </contrib-group>
      <pub-date>
        <year>2024</year>
      </pub-date>
      <fpage>26</fpage>
      <lpage>29</lpage>
      <abstract>
        <p>1. Summary of the project</p>
      </abstract>
    </article-meta>
  </front>
  <body>
    <sec id="sec-1">
      <title>-</title>
      <p>Plants possess an intricate chemo-diversity, serving as a rich reservoir for the discovery of potential
therapeutic agents. In the framework of a Swiss research initiative called "Sinergia", six research
groups from diferent disciplines collaborate to explore the potential of more than 17,000 unique dried
plant extracts, to uncover novel bioactive molecules. To do so, heterogenous data from the diferent
research groups was integrated in a Knowledge Graph (KG). This includes (i) highresolution mass
spectrometry data, (ii) taxonomical information, (iii) chemo-informatics results, (iv) bioassay outcomes
for tuberculosis, obesity, anticancer and antiviral models, as well as (v) data from organic synthetic
chemistry.</p>
      <p>We wanted to create a comprehensive framework to facilitate the alignment of multimodal data
across chemical structures, biological activities, spectral features, and taxonomy, among others. At its
core, it would include experimental data that is processed at the sample level, harmonized with external
identifiers whenever feasible, semantically enriched and integrated into the KG.</p>
      <p>The KG allows to enhance eficient data mining capabilities to address diferent scientific research
questions related to the specific objectives of each participating group. For example, "Which bioactive
molecules annotated in the extract of ’plant A’ are present in other species of the same genus?", “Which
taxonomic group presents a remarkable concentration of bioactive molecules for a specific assay?", or
"Which non-toxic active extracts (for a specific assay) present a high probability of containing new
molecules?".</p>
      <p>
        The modelling of data from chemical analysis became a subproject in itself: ENPKG [
        <xref ref-type="bibr" rid="ref1">1</xref>
        ]. The recently
published paper describing it exemplifies how this KG can be used to answer questions centered around
the chemistry of incompletely described natural products.
      </p>
      <p>The heterogeneity, expansion, and evolution of the data throughout time posed a major challenge for
the project, centered on managing, integrating, modeling, and efectively sharing data generated by the
diferent groups. Also, compliance with a data management plan and adherence to open-source science
principles, particularly the FAIR (Findable, Accessible, Interoperable, Reusable) principles, had to be
ensured.</p>
      <p>
        To overcome these challenges and handle this intricate data, a custom tool called KGSteward [
        <xref ref-type="bibr" rid="ref2">2</xref>
        ] was
developed. This tool enables content synchronization based on a centrally managed versioncontrolled
configuration file using Git [
        <xref ref-type="bibr" rid="ref3">3</xref>
        ]. This strategic approach provides the flexibility required to address the
global project challenges in efectively managing shared data.
      </p>
      <p>
        For managing the KG, the free version of GraphDB [
        <xref ref-type="bibr" rid="ref4">4</xref>
        ] was chosen, to which KGSteward is connected.
KGSteward updates the KG through the Unix command line and relies on a YAML configuration file
shared through Git, enabling users to host their local instances of the triplestore. The primary functions
performed by KGSteward include: (i) creating the repository, (ii) uploading the TTL files listed in the
configuration file, and (iii) executing UPDATE queries to clean and harmonize the KG.
      </p>
      <p>Table 1 provides a snapshot of the diferent datasets integrated into the KG, emphasizing their origin.
These datasets, presented as RDF graphs, are organized around the central concept of an "analysis run".
This documents data executed by the same operator within the same laboratory on a specific date.
Should the operator repeat the same assay on another date, it is regarded as a separate analysis run to
ensure comprehensive traceability of integrated data.</p>
      <p>
        Provenance information is encoded using the PROV Ontology [
        <xref ref-type="bibr" rid="ref5">5</xref>
        ], linking to raw data files, operators,
protocols, and associated articles. This ensures the ‘Findable and Reusable’ part of the FAIRification.
Additionally, for Interoperability, vocabularies like RDF, RDFS, OWL, are employed to describe data and
metadata extensively, as described in [6]. We also created a customized vocabulary for project concepts
that couldn’t be mapped on existing vocabularies. It is planned to progressively release the data in the
public domain at the same time as the scientific results’ publications. Currently the KG contains about
200 million triples.
      </p>
      <p>The use of RDF/SPARQL in our project has proven to be highly robust, particularly in handling the
dynamic nature of our evolving datasets. These semantic technologies ofer a notable level of flexibility
in defining and adapting vocabularies according to our project’s specific requirements.</p>
      <p>However, the delicate balance between flexibility and usability was a central concern during the
deployment of our KG. The KG structure was modelled from a starting set of data, but due to the quick
evolution of the cumulative heterogenous sets it was modified drastically for a better overall fit. The
implementation of KGSteward in the evolution process was instrumental, as it bypasses the graphical
interface, automatically detects the modified input TTL files, and does selective updates.</p>
      <p>This strategic approach reflects the project’s dedication to efectively capture, organize, and adapt to
the continuously generated data and concepts. The integration of RDF/SPARQL aligns with our project
goals and reflects our commitment to navigate the complexities of contemporary data management
challenges in interdisciplinary research endeavors.</p>
    </sec>
    <sec id="sec-2">
      <title>2. Acknowledgements</title>
      <p>The authors are grateful to Green Mission Pierre Fabre, Pierre Fabre Research Institute, Toulouse, France,
for establishing and sharing this unique library of extracts. The authors thank the Swiss National
Science Foundation for received support for the project (SNF N° CRSII5_189921/1).</p>
    </sec>
    <sec id="sec-3">
      <title>3. References</title>
      <p>S. Zednik, J. Zhao, PROV-O: The PROV Ontology, World Wide Web Consortium, United States,
2013.
[6] M. Dumontier, A.J.G. Gray, M.S. Marshall, V. Alexiev, P. Ansell, G. Bader, J. Baran, J.T. Bolleman, A.</p>
      <p>Callahan, J. Cruz-Toledo, P. Gaudet, E.A. Gombocz, A.N. Gonzalez-Beltran, P. Groth, M. Haendel,
M. Ito, S. Jupp, N. Juty, T. Katayama, N. Kobayashi, K. Krishnaswami, C. Laibe, N. Le Novère,
S. Lin, J. Malone, M. Miller, C.J. Mungall, L. Rietveld, S.M. Wimalaratne, A. Yamaguchi, The
health care and life sciences community profile for dataset descriptions, PeerJ. 4 (2016) e2331.
doi:10.7717/peerj.2331.
[7] L.-M. Quiros-Guerrero, L.-F. Nothias, A. Gaudry, L. Marcourt, P.-M. Allard, A. Rutz, B.</p>
      <p>David, E.F. Queiroz, J.-L. Wolfender, Inventa: A computational tool to discover structural
novelty in natural extracts libraries, Frontiers in Molecular Biosciences. 9 (2022) 1028334.
doi:10.3389/fmolb.2022.1028334.
[8] P.-M. Allard, A. Gaudry, L.-M. Quirós-Guerrero, A. Rutz, M. Dounoue-Kubo, T.W.N. Walker, E.</p>
      <p>Defossez, C. Long, A. Grondin, B. David, J.-L. Wolfender, Open and reusable annotated mass
spectrometry dataset of a chemodiverse collection of 1,600 plant extracts, GigaScience. 12 (2022)
giac124. doi:10.1093/gigascience/giac124.
[9] A. Rutz, M. Sorokina, J. Galgonek, D. Mietchen, E. Willighagen, A. Gaudry, J.G. Graham, R. Stephan,
R. Page, J. Vondrášek, C. Steinbeck, G.F. Pauli, J.-L. Wolfender, J. Bisson, P.-M. Allard, The LOTUS
initiative for open knowledge management in natural products research, ELife. 11 (2022) e70780.
doi:10.7554/eLife.70780.</p>
    </sec>
  </body>
  <back>
    <ref-list>
      <ref id="ref1">
        <mixed-citation>
          [1]
          <string-name>
            <given-names>A.</given-names>
            <surname>Gaudry</surname>
          </string-name>
          ,
          <string-name>
            <given-names>M.</given-names>
            <surname>Pagni</surname>
          </string-name>
          ,
          <string-name>
            <given-names>F.</given-names>
            <surname>Mehl</surname>
          </string-name>
          ,
          <string-name>
            <given-names>S.</given-names>
            <surname>Moretti</surname>
          </string-name>
          , L.
          <string-name>
            <surname>-M. Quiros-Guerrero</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          <string-name>
            <surname>Rutz</surname>
            ,
            <given-names>M.</given-names>
          </string-name>
          <string-name>
            <surname>Kaiser</surname>
            ,
            <given-names>L.</given-names>
          </string-name>
          <string-name>
            <surname>Marcourt</surname>
            ,
            <given-names>E. Ferreira</given-names>
          </string-name>
          <string-name>
            <surname>Queiroz</surname>
            ,
            <given-names>J.-R.</given-names>
          </string-name>
          <string-name>
            <surname>Ioset</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          <string-name>
            <surname>Grondin</surname>
            ,
            <given-names>B.</given-names>
          </string-name>
          <string-name>
            <surname>David</surname>
            ,
            <given-names>J.-L.</given-names>
          </string-name>
          <string-name>
            <surname>Wolfender</surname>
          </string-name>
          ,
          <string-name>
            <surname>P.-M. Allard</surname>
            ,
            <given-names>A SampleCentric</given-names>
          </string-name>
          and
          <article-title>Knowledge-Driven Computational Framework for Natural Products Drug Discovery</article-title>
          , Chemistry,
          <year>2023</year>
          . doi:
          <volume>10</volume>
          .26434/chemrxiv-2023-sljbt.
        </mixed-citation>
      </ref>
      <ref id="ref2">
        <mixed-citation>
          [2]
          <string-name>
            <given-names>M.</given-names>
            <surname>Pagni</surname>
          </string-name>
          , KGSteward,
          <year>2023</year>
          . https://github.com/sib-swiss/kgsteward.
        </mixed-citation>
      </ref>
      <ref id="ref3">
        <mixed-citation>
          [3]
          <string-name>
            <given-names>S.</given-names>
            <surname>Chacon</surname>
          </string-name>
          ,
          <string-name>
            <given-names>B.</given-names>
            <surname>Straub</surname>
          </string-name>
          , Pro git,
          <year>2014</year>
          . https://git-scm.com/.
        </mixed-citation>
      </ref>
      <ref id="ref4">
        <mixed-citation>
          [4]
          <string-name>
            <surname>Ontotext</surname>
            ,
            <given-names>GraphDB</given-names>
          </string-name>
          ,
          <year>2023</year>
          . https://www.ontotext.com/products/graphdb/.
        </mixed-citation>
      </ref>
      <ref id="ref5">
        <mixed-citation>
          [5]
          <string-name>
            <given-names>T.</given-names>
            <surname>Lebo</surname>
          </string-name>
          ,
          <string-name>
            <given-names>S.</given-names>
            <surname>Sahoo</surname>
          </string-name>
          ,
          <string-name>
            <given-names>D.</given-names>
            <surname>McGuinness</surname>
          </string-name>
          ,
          <string-name>
            <given-names>K.</given-names>
            <surname>Belhajjame</surname>
          </string-name>
          ,
          <string-name>
            <given-names>J.</given-names>
            <surname>Cheney</surname>
          </string-name>
          ,
          <string-name>
            <given-names>D.</given-names>
            <surname>Corsar</surname>
          </string-name>
          ,
          <string-name>
            <given-names>D.</given-names>
            <surname>Garijo</surname>
          </string-name>
          ,
          <string-name>
            <given-names>S.</given-names>
            <surname>Soiland-Reyes</surname>
          </string-name>
          ,
        </mixed-citation>
      </ref>
    </ref-list>
  </back>
</article>