<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD v1.0 20120330//EN" "JATS-archivearticle1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink">
  <front>
    <journal-meta>
      <journal-title-group>
        <journal-title>Journal of Biomedical Semantics 8
(2017) 56:1-56:28. URL: https://doi.org/10.1186/s13326</journal-title>
      </journal-title-group>
    </journal-meta>
    <article-meta>
      <article-id pub-id-type="doi">10.1093/database/baac035</article-id>
      <title-group>
        <article-title>Results of the Ontology Alignment Evaluation Initiative 2023</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="author">
          <string-name>Mina Abd Nikooie Pour</string-name>
          <xref ref-type="aff" rid="aff19">19</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Alsayed Algergawy</string-name>
          <xref ref-type="aff" rid="aff12">12</xref>
          <xref ref-type="aff" rid="aff4">4</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Patrice Buche</string-name>
          <xref ref-type="aff" rid="aff23">23</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Leyla J. Castro</string-name>
          <xref ref-type="aff" rid="aff26">26</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Jiaoyan Chen</string-name>
          <xref ref-type="aff" rid="aff7">7</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Adrien Coulet</string-name>
          <xref ref-type="aff" rid="aff14">14</xref>
          <xref ref-type="aff" rid="aff3">3</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Julien Cufi</string-name>
          <xref ref-type="aff" rid="aff23">23</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Hang Dong</string-name>
          <xref ref-type="aff" rid="aff8">8</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Omaima Fallatah</string-name>
          <xref ref-type="aff" rid="aff9">9</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Daniel Faria</string-name>
          <xref ref-type="aff" rid="aff13">13</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Irini Fundulaki</string-name>
          <xref ref-type="aff" rid="aff17">17</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Sven Hertling</string-name>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Yuan He</string-name>
          <xref ref-type="aff" rid="aff8">8</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Ian Horrocks</string-name>
          <xref ref-type="aff" rid="aff8">8</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Martin Huschka</string-name>
          <xref ref-type="aff" rid="aff11">11</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Liliana Ibanescu</string-name>
          <xref ref-type="aff" rid="aff24">24</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Sarika Jain</string-name>
          <xref ref-type="aff" rid="aff20">20</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Ernesto Jime´nez-Ruiz</string-name>
          <xref ref-type="aff" rid="aff5">5</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Naouel Karam</string-name>
          <xref ref-type="aff" rid="aff16">16</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Patrick Lambrix</string-name>
          <xref ref-type="aff" rid="aff19">19</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Huanyu Li</string-name>
          <xref ref-type="aff" rid="aff19">19</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Ying Li</string-name>
          <xref ref-type="aff" rid="aff19">19</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Pierre Monnin</string-name>
          <xref ref-type="aff" rid="aff25">25</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Engy Nasr</string-name>
          <xref ref-type="aff" rid="aff1">1</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Heiko Paulheim</string-name>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Catia Pesquita</string-name>
          <xref ref-type="aff" rid="aff18">18</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Tzanina Saveta</string-name>
          <xref ref-type="aff" rid="aff17">17</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Pavel Shvaiko</string-name>
          <xref ref-type="aff" rid="aff22">22</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Guilherme Sousa</string-name>
          <xref ref-type="aff" rid="aff15">15</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Cassia Trojahn</string-name>
          <xref ref-type="aff" rid="aff15">15</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Jana Vatascinova</string-name>
          <xref ref-type="aff" rid="aff21">21</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Mingfang Wu</string-name>
          <xref ref-type="aff" rid="aff2">2</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Beyza Yaman</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Ondrej Zamazal</string-name>
          <xref ref-type="aff" rid="aff21">21</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Lu Zhou</string-name>
          <xref ref-type="aff" rid="aff10">10</xref>
        </contrib>
        <aff id="aff0">
          <label>0</label>
          <institution>ADAPT Centre, Trinity College Dublin</institution>
        </aff>
        <aff id="aff1">
          <label>1</label>
          <institution>Albert Ludwig University of Freiburg</institution>
          ,
          <country country="DE">Germany</country>
        </aff>
        <aff id="aff2">
          <label>2</label>
          <institution>Australian Research Data Commons</institution>
        </aff>
        <aff id="aff3">
          <label>3</label>
          <institution>Centre de Recherche des Cordeliers, Inserm, Universite ́ Paris Cite ́, Sorbonne Universite ́</institution>
          ,
          <country country="FR">France</country>
        </aff>
        <aff id="aff4">
          <label>4</label>
          <institution>Chair of Data and Knowledge Engineering, University of Passau</institution>
          ,
          <country country="DE">Germany</country>
        </aff>
        <aff id="aff5">
          <label>5</label>
          <institution>City, University of London, UK &amp; SIRIUS, University of Oslo</institution>
          ,
          <country country="NO">Norway</country>
        </aff>
        <aff id="aff6">
          <label>6</label>
          <institution>Data and Web Science Group, University of Mannheim</institution>
          ,
          <country country="DE">Germany</country>
        </aff>
        <aff id="aff7">
          <label>7</label>
          <institution>Department of Computer Science, The University of Manchester</institution>
          ,
          <country country="UK">UK</country>
        </aff>
        <aff id="aff8">
          <label>8</label>
          <institution>Department of Computer Science, University of Oxford</institution>
          ,
          <country country="UK">UK</country>
        </aff>
        <aff id="aff9">
          <label>9</label>
          <institution>Department of Data Science, Umm Al-Qura University</institution>
          ,
          <country country="SA">Saudi Arabia</country>
        </aff>
        <aff id="aff10">
          <label>10</label>
          <institution>Flatfee Corp</institution>
          ,
          <country country="US">USA</country>
        </aff>
        <aff id="aff11">
          <label>11</label>
          <institution>Fraunhofer Institute for High-Speed Dynamics, Ernst-Mach-Institut</institution>
          ,
          <addr-line>EMI</addr-line>
          ,
          <country country="DE">Germany</country>
        </aff>
        <aff id="aff12">
          <label>12</label>
          <institution>Heinz Nixdorf Chair for Distributed Information Systems, Friedrich Schiller University Jena</institution>
          ,
          <country country="DE">Germany</country>
        </aff>
        <aff id="aff13">
          <label>13</label>
          <institution>INESC-ID / IST, University of Lisbon</institution>
          ,
          <country country="PT">Portugal</country>
        </aff>
        <aff id="aff14">
          <label>14</label>
          <institution>Inria Paris</institution>
          ,
          <country country="FR">France</country>
        </aff>
        <aff id="aff15">
          <label>15</label>
          <institution>Institut de Recherche en Informatique de Toulouse</institution>
          ,
          <country country="FR">France</country>
        </aff>
        <aff id="aff16">
          <label>16</label>
          <institution>Institute for Applied Informatics, University of Leipzig</institution>
          ,
          <country country="DE">Germany</country>
        </aff>
        <aff id="aff17">
          <label>17</label>
          <institution>Institute of Computer Science-FORTH</institution>
          ,
          <addr-line>Heraklion</addr-line>
          ,
          <country country="GR">Greece</country>
        </aff>
        <aff id="aff18">
          <label>18</label>
          <institution>LASIGE, Faculdade de Cieˆncias, Universidade de Lisboa</institution>
          ,
          <country country="PT">Portugal</country>
        </aff>
        <aff id="aff19">
          <label>19</label>
          <institution>Linko ̈ping University &amp; Swedish e-Science Research Centre</institution>
          ,
          <addr-line>Linko ̈ping</addr-line>
          ,
          <country country="SE">Sweden</country>
        </aff>
        <aff id="aff20">
          <label>20</label>
          <institution>National Institute of Technology Kurukshetra</institution>
          ,
          <country country="IN">India</country>
        </aff>
        <aff id="aff21">
          <label>21</label>
          <institution>Prague University of Economics and Business</institution>
          ,
          <country country="CZ">Czech Republic</country>
        </aff>
        <aff id="aff22">
          <label>22</label>
          <institution>Trentino Digitale SpA</institution>
          ,
          <addr-line>Trento</addr-line>
          ,
          <country country="IT">Italy</country>
        </aff>
        <aff id="aff23">
          <label>23</label>
          <institution>UMR IATE, INRAE, University of Montpellier</institution>
          ,
          <country country="FR">France</country>
        </aff>
        <aff id="aff24">
          <label>24</label>
          <institution>Universite ́ Paris-Saclay</institution>
          ,
          <addr-line>INRAE, AgroParisTech, UMR MIA Paris-Saclay</addr-line>
          ,
          <country country="FR">France</country>
        </aff>
        <aff id="aff25">
          <label>25</label>
          <institution>University Coˆte d'Azur</institution>
          ,
          <addr-line>Inria, CNRS, I3S</addr-line>
          ,
          <country country="FR">France</country>
        </aff>
        <aff id="aff26">
          <label>26</label>
          <institution>ZB MED Information Centre for Life Sciences</institution>
          ,
          <country country="DE">Germany</country>
        </aff>
      </contrib-group>
      <pub-date>
        <year>2022</year>
      </pub-date>
      <volume>3324</volume>
      <fpage>1</fpage>
      <lpage>56</lpage>
      <abstract>
        <p>The Ontology Alignment Evaluation Initiative (OAEI) aims at comparing ontology matching systems on precisely defined test cases. These test cases can be based on ontologies of different levels of complexity and use different evaluation modalities. The OAEI 2023 campaign offered 15 tracks and was attended by 16 participants. This paper is an overall presentation of that campaign.</p>
      </abstract>
    </article-meta>
  </front>
  <body>
    <sec id="sec-1">
      <title>1. Introduction</title>
      <p>
        The Ontology Alignment Evaluation Initiative1 (OAEI) is a coordinated international initiative,
which organizes the evaluation of ontology matching systems [
        <xref ref-type="bibr" rid="ref1 ref2">1, 2</xref>
        ], and which has been run for
eighteen years now. The main goal of the OAEI is to compare systems and algorithms openly and
on the same basis to allow anyone to conclude the best ontology matching strategies. Furthermore,
the ambition is that, from such evaluations, developers can improve their systems and offer better
tools addressing the evolving application needs.
      </p>
      <p>
        Two first events were organized in 2004: (i) the Information Interpretation and Integration
Conference (I3CON) held at the NIST Performance Metrics for Intelligent Systems (PerMIS)
workshop and (ii) the Ontology Alignment Contest held at the Evaluation of Ontology-based
Tools (EON) workshop of the annual International Semantic Web Conference (ISWC) [
        <xref ref-type="bibr" rid="ref3">3</xref>
        ]. Then,
a unique OAEI campaign occurred in 2005 at the workshop on Integrating Ontologies held in
conjunction with the International Conference on Knowledge Capture (K-Cap) [
        <xref ref-type="bibr" rid="ref4">4</xref>
        ]. From 2006
until the present, the OAEI campaigns were held at the Ontology Matching workshop, co-located
with ISWC [
        <xref ref-type="bibr" rid="ref5 ref6">5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21</xref>
        ], which this year took place
in Athens, Greece2.
      </p>
      <p>Since 2011, we have been using an environment for automatically processing evaluations
which was developed within the SEALS (Semantic Evaluation At Large Scale) project3. SEALS
provided a software infrastructure for automatically executing evaluations and evaluation
campaigns for typical semantic web tools, including ontology matching. Since OAEI 2017, a novel
evaluation environment called HOBBIT (Section 2.1) was adopted for the HOBBIT Link
Discovery track, and later extended to enable the evaluation of other tracks. Some tracks are run
exclusively through SEALS and others through HOBBIT, but several allow participants to choose
their preferred platform. Since last year, the MELT framework [22] has been adopted to facilitate
the SEALS and HOBBIT wrapping and evaluation. This year, most tracks have adopted MELT
as their evaluation platform.</p>
      <p>This paper synthesizes the 2023 evaluation campaign and introduces the results provided in the
participants’ papers. The remainder of the paper is organized as follows: in Section 2, we present
the overall evaluation methodology; in Section 3, we present the tracks and datasets; in Section 4
we present and discuss the results; and finally, Section 5 discusses the lessons learned.
OM 2023: The 18th International Workshop on Ontology Matching collocated with the 22nd International Semantic
Web Conference ISWC-2023 November 7th, 2023, Athens, Greece</p>
      <p>© 2023 Copyright for this paper by its authors. Use permitted under Creative Commons License Attribution 4.0 International (CC BY 4.0).</p>
      <p>CPWrEooUrckResehdoinpgs IhStpN:/c1e6u1r3-w-0s.o7r3g CEUR Workshop Proceedings (CEUR-WS.org)
1http://oaei.ontologymatching.org
2http://om2023.ontologymatching.org
3http://www.seals-project.eu</p>
    </sec>
    <sec id="sec-2">
      <title>2. Methodology</title>
      <sec id="sec-2-1">
        <title>2.1. Evaluation platforms</title>
        <p>The OAEI evaluation was conducted in one of three alternative platforms: the SEALS client, the
HOBBIT platform, or the MELT framework. All of them have the goal of ensuring reproducibility
and comparability of the results across matching systems. As of this campaign, the use of the
SEALS client and packaging format is deprecated in favor of MELT, with the sole exception of
the Interactive Matching track, as simulated interactive matching is not yet supported by MELT.</p>
        <p>The SEALS client was developed in 2011. It is a Java-based command line interface for
ontology matching evaluation, which requires system developers to implement an interface and
to wrap their tools in a predefined way, including all required libraries and resources.</p>
        <p>The HOBBIT platform4 was introduced in 2017. It is a web interface for linked data and
ontology matching evaluation, which requires systems to be wrapped inside docker containers
and includes a SystemAdapter class, then being uploaded into the HOBBIT platform [23].</p>
        <p>The MELT framework5 [22] was introduced in 2019 and is under active development. It
allows the development, evaluation, and packaging of matching systems for evaluation interfaces
like SEALS or HOBBIT. It further enables developers to use Python or any other programming
language in their matching systems, which beforehand had been a hurdle for OAEI participants.
The evaluation client6 allows organizers to evaluate packaged systems whereby multiple
submission formats are supported (SEALS packages or matchers implemented as Web services).
Starting with this year, the MELT framework also supports the SSSOM [24] format. Therefore,
systems producing an alignment in the SSSOM format can be evaluated as well.</p>
        <p>All platforms compute the standard evaluation metrics against the reference alignments:
precision, recall, and F-measure. In test cases requiring different evaluation modalities, evaluation was
carried out a posteriori, using the alignments produced by the matching systems.</p>
      </sec>
      <sec id="sec-2-2">
        <title>2.2. Submission formats</title>
        <p>This year, three submission formats were allowed: (1) SEALS package, (2) HOBBIT, and (3)
MELT Web interface. With the increasing usage of other programming languages than Java and
increasing hardware requirements for matching systems, since 2021 the MELT Web interface was
introduced to address this issue. It mainly consists of a technology-independent HTTP interface7
which participants can implement as they wish. Alternatively, they can use the MELT framework
to assist them, as it can be used to wrap any matching system as docker container implementing
the HTTP interface. In 2023, 12 systems were submitted as MELT Web docker container, 3
systems were submitted as SEALS package, 1 system was uploaded to the HOBBIT platform,
and one system implemented the Web interface directly and provided hosting for the system.</p>
        <p>In this year we also allowed to submit alignment files in addition to the executable system in
case it requires substantial hardware or software resources.</p>
        <sec id="sec-2-2-1">
          <title>4https://project-hobbit.eu/outcomes/hobbit-platform/ 5https://github.com/dwslab/melt 6https://dwslab.github.io/melt/matcher-evaluation/client 7https://dwslab.github.io/melt/matcher-packaging/web</title>
        </sec>
      </sec>
      <sec id="sec-2-3">
        <title>2.3. OAEI campaign phases</title>
        <p>As in previous years, the OAEI 2023 campaign was divided into three phases: preparatory,
execution, and evaluation.</p>
        <p>In the preparatory phase, the test cases were provided to participants in an initial assessment
period between June 30ℎ and July 31, 2023. The goal of this phase is to ensure that the test
cases make sense to participants, and give them the opportunity to provide feedback to organizers
on the test case as well as potentially report errors. At the end of this phase, the final test base
was frozen and released.</p>
        <p>During the ensuing execution phase, participants test and potentially develop their matching
systems to automatically match the test cases. Participants can self-evaluate their results either
by comparing their output with the reference alignments or by using either of the evaluation
platforms. They can tune their systems with respect to the non-blind evaluation as long as they
respect the rules of the OAEI. Participants were required to register their systems by July 31
and make a preliminary evaluation by August 31. The execution phase was terminated on
September 30ℎ, 2023, at which date participants had to submit the (near) final versions of their
systems (SEALS-wrapped and/or HOBBIT-wrapped).</p>
        <p>During the evaluation phase, systems were evaluated by all track organizers. In case minor
problems were found during the initial stages of this phase, they were reported to the developers,
who were given the opportunity to fix and resubmit their systems. Initial results were provided
directly to the participants, whereas final results for most tracks were published on the respective
OAEI web pages before the workshop.</p>
      </sec>
    </sec>
    <sec id="sec-3">
      <title>3. Tracks and Test Cases</title>
      <p>This year’s OAEI campaign consisted of 15 tracks, all of them including OWL ontologies while
only one also including SKOS thesauri, namely the Biodiversity and the Ecology track. They can
be grouped into:
– Schema matching tracks, which have as objective matching ontology classes and/or
properties.
– Instance matching tracks, which have as objective matching ontology instances.
– Instance and schema matching tracks, which involve both of the above.
– Complex matching tracks, which have as objective finding complex correspondences
between ontology entities.
– Interactive tracks, which simulate user interaction to enable the benchmarking of interactive
matching algorithms.</p>
      <p>The tracks are summarized in Table 1 and detailed in the following sections.</p>
      <p>formalism
relations
confidence modalities language SEALS HOBBIT MELT
3.1. Anatomy</p>
      <p>=
=, &lt;=
=
=
=, &lt;=
=, &lt;=
=, &lt;=</p>
      <p>=
=, &lt;=, &gt;=
=
=
=
=
=
=, &lt;, &gt;,</p>
      <p>Close, Related
The anatomy track comprises a single test case consisting of matching two fragments of
biomedical ontologies which describe the human anatomy8 (3304 classes) and the anatomy of the mouse9
(2744 classes). The evaluation is based on a manually curated reference alignment. This dataset
has been used since 2007 with some improvements over the years [25].</p>
      <p>Systems are evaluated with the standard parameters of precision, recall, F-measure.
Additionally, recall+ is computed by excluding trivial correspondences (i.e., correspondences that
have the same normalized label). Alignments are also checked for coherence using the Pellet
reasoner. The evaluation was carried out on a machine with a 5 core CPU @ 1.80 GHz with
16GB allocated RAM, using the MELT framework. For some systems, the SEALS client has
been used. However, the evaluation parameters were computed a posteriori, after removing from
the alignments produced by the systems, correspondences expressing relations other than
equivalence, as well as trivial correspondences in the oboInOwl namespace (e.g., oboInOwl#Synonym
= oboInOwl#Synonym). The results obtained with the SEALS client vary in some cases by 0.5%
compared to the results presented in Section 4.2.</p>
      <sec id="sec-3-1">
        <title>3.2. Conference</title>
        <p>The conference track feature two test cases. The main test case is a suite of 21 matching tasks
corresponding to the pairwise combination of 7 moderately expressive ontologies describing the</p>
        <sec id="sec-3-1-1">
          <title>8https://www.cancer.gov/cancertopics/cancerlibrary/terminologyresources</title>
          <p>9http://www.informatics.jax.org/searches/AMA form.shtml
domain of organizing conferences. The dataset and its usage are described in [26]. This year we
again run a second test case consisting of a suite of three tasks of matching DBpedia ontology
(filtered to the dbpedia namespace) and three ontologies from the conference domain.</p>
          <p>For the main test case the track uses several reference alignments for evaluation: the old (and
not fully complete) manually curated open reference alignment, ra1; an extended, also manually
curated version of this alignment, ra2; a version of the latter corrected to resolve violations of
conservativity, rar2; and an uncertain version of ra1 produced through crowd-sourcing, where
the score of each correspondence is the fraction of people in the evaluation group that agree
with the correspondence. The latter reference was used in two evaluation modalities: discrete
and continuous evaluation. In the former, correspondences in the uncertain reference alignment
with a score of at least 0.5 are treated as correct whereas those with lower score are treated
as incorrect, and standard evaluation parameters are used to evaluated systems. In the latter,
weighted precision, recall and F-measure values are computed by taking into consideration the
actual scores of the uncertain reference, as well as the scores generated by the matching system.
For the sharp reference alignments (ra1, ra2 and rar2), the evaluation is based on the standard
parameters, as well the F0.5-measure and F2-measure and on conservativity and consistency
violations. Whereas F1 is the harmonic mean of precision and recall where both receive equal
weight, 2 gives higher weight to recall than precision and F0.5 gives higher weight to precision
higher than recall. The second test case contains open reference alignment and systems were
evaluated using the standard metrics.</p>
          <p>Two baseline matchers are used to benchmark the systems: edna string edit distance matcher;
and StringEquiv string equivalence matcher as in the anatomy test case.</p>
        </sec>
      </sec>
      <sec id="sec-3-2">
        <title>3.3. Multifarm</title>
        <p>The multifarm track [27] aims at evaluating the ability of matching systems to deal with ontologies
in different natural languages. This dataset results from the translation of 7 ontologies from the
conference track (cmt, conference, confOf, iasted, sigkdd, ekaw and edas) into 10 languages:
Arabic (ar), Chinese (cn), Czech (cz), Dutch (nl), French (fr), German (de), Italian (it), Portuguese
(pt), Russian (ru), and Spanish (es). The dataset is composed of 55 pairs of languages, with 49
matching tasks for each of them, taking into account the alignment direction (e.g. cmt →edas
and cmt →edas are distinct matching tasks). While part of the dataset is openly available, all
matching tasks involving the edas and ekaw ontologies (resulting in 55 × 24 matching tasks) are
used for blind evaluation.</p>
        <p>We consider two test cases: i) those tasks where two different ontologies (cmt→edas, for
instance) have been translated into two different languages; and ii) those tasks where the same
ontology (cmt→cmt) has been translated into two different languages. For the tasks of type ii),
good results are not only related to the use of specific techniques for dealing with cross-lingual
ontologies, but also on the ability to exploit the identical structure of the ontologies. This year,
we report the results on different ontologies (i).</p>
        <p>The reference alignments used in this track derive directly from the manually curated
Conference ra1 reference alignments. In 2021, alignments have been manually evaluated by domain
experts. The evaluation is blind. The systems have been executed on a Ubuntu Linux machine
configured with 32GB of RAM running under a Intel Core CPU 2.00GHz x8 cores. The
evaluation was performed using the MELT platform. Every participating system was executed in its
standard setting and we compare precision, recall and F-measure as well as the computation time.</p>
      </sec>
      <sec id="sec-3-3">
        <title>3.4. Complex Matching</title>
        <p>The complex matching track is meant to evaluate the matchers based on their ability to
generate complex alignments. A complex alignment is composed of complex correspondences
typically involving more than two ontology entities, such as 1:AcceptedPaper ≡ 2:Paper ⊓
2:hasDecision.2:Acceptance.</p>
        <p>This year the track run with two data sets from the conference domain: Conference and
Populated Conference, as the other complex sub-tracks (Hydrography, GeoLink, Populated
GeoLink Populated Enslaved, and Taxon datasets) have been discontinued.</p>
        <p>The Conference dataset comprises three ontologies: cmt, conference, and ekaw from the
conference dataset. The reference alignment was created as a consensus between experts. To
allow matchers which rely on instances to participate over the Conference complex track, the
Populated Conference data set is composed of 5 conference ontologies populated with more or
less common instances, resulting in 6 datasets: (6 versions on the repository: v0, v20, v40, v60,
v80 and v100). Details on the population and evaluation modalities are available10.</p>
        <p>The systems have been executed on a Ubuntu Linux machine configured with 32GB of RAM
running under a Intel Core CPU 2.00GHz x8 processors.
3.5. Food
The Food Nutritional Composition track aims at finding alignments between food concepts
from CIQUAL11, the French food nutritional composition database, and food concepts from
SIREN12, the Scientific Information and Retrieval Exchange Network of the US Food and Drug
administration. Foods from both databases are described in LanguaL13, a well-known multilingual
thesaurus using faceted classification. LanguaL stands for “Langua aLimentaria” or “language of
food”; more than 40,000 foods used in food composition databases are described using LanguaL.</p>
        <p>In [28], a method to provide OWL modelling of food concepts from both datasets, CIQUAL14
and SIREN 15, and a gold standard are presented.</p>
        <p>The evaluation was performed using the MELT platform. Every participating system was
executed in its standard setting and we compare precision, recall and F-measure as well as the
computation time.</p>
      </sec>
      <sec id="sec-3-4">
        <title>3.6. Interactive Matching</title>
        <p>The interactive matching track aims to assess the performance of semi-automated matching
systems by simulating user interaction [29, 30, 31]. The evaluation thus focuses on how interaction
10https://framagit.org/IRIT UT2J/conference-dataset-population
11https://ciqual.anses.fr/
12http://langual.org/langual indexed datasets.asp
13https://www.langual.org/default.asp
14https://entrepot.recherche.data.gouv.fr/dataset.xhtml?persistentId=doi:10.15454/6CEYU3
15https://entrepot.recherche.data.gouv.fr/dataset.xhtml?persistentId=doi:10.15454/5LLGVY
with the user improves the matching results. Currently, this track does not evaluate the user
experience or the user interfaces of the systems [32, 30].</p>
        <p>The interactive matching track is based on the datasets from the Anatomy and Conference tracks,
which have been previously described. It relies on the SEALS client’s Oracle class to simulate
user interactions. An interactive matching system can present a collection of correspondences
simultaneously to the oracle, telling the system whether that correspondence is correct or not.
If a system presents up to three correspondences together and each correspondence presented
has a mapped entity (i.e., class or property) in common with at least one other correspondence
presented, the oracle counts this as a single interaction, under the rationale that this corresponds
to a scenario where a user is asked to choose between conflicting candidate correspondences. To
simulate the possibility of user errors, the oracle can be set to reply with a given error probability
(randomly, from a uniform distribution). We evaluated systems with four different error rates: 0.0
(perfect user), 0.1, 0.2, and 0.3.</p>
        <p>In addition to the standard evaluation parameters, we also compute the number of requests
made by the system, the total number of distinct correspondences asked, the number of positive
and negative answers from the oracle, the performance of the system according to the oracle (to
assess the impact of the oracle errors on the system) and finally, the performance of the oracle
itself (to assess how erroneous it was).</p>
        <p>The evaluation was carried out on a server with 3.46 GHz (6 cores) and 8GB RAM allocated
to the matching systems. For systems requiring more RAM, the evaluation was carried out on a
computer with an AMD Ryzen 7 5700G 3.80 GHz CPU and 32GB RAM, with 10GB of max
heap space allocated to java.Each system was run ten times and the final result of a system for
each error rate represents the average of these runs. For the Conference dataset with the ra1
alignment, precision and recall correspond to the micro-average over all ontology pairs, whereas
the number of interactions is the total number of interactions for all the pairs.
3.7. Bio-ML
The Bio-ML track [33] incorporates both equivalence and subsumption ontology matching
(OM) tasks for biomedical ontologies, with ground truth (equivalence) mappings extracted
from Mondo [34] and UMLS [35] (see Table 2). Mondo aims to integrate disease concepts
worldwide, while UMLS is a meta-thesaurus for the biomedical domain. Based on techniques
(ontology pruning, subsumption mapping construction, negative candidate mapping generation,
etc.) proposed in [33], we introduced vfie OM pairs with their information reported in Table 3.
Each OM pair is accompanied with both equivalence and subsumption matching tasks; each
matching task has two data split settings, i.e., unsupervised setting with no training mappings,
and semi-supervised setting with 30% ground truth mappings for training/validation. In the 2023
edition, we made several significant updates:
– Logical module enrichment: we adopted locality-based logical modules [36] to enrich the
existing pruned ontologies to provide more contexts for alignment; the added entities are
annotated as “not used in alignment” and will be ignored in evaluation.
– Bio-LLM sub-track: To support more efficient evaluation of large language model-based
OM, we introduced a special sub-track called Bio-LLM [37], which consists of challenging
subsets of NCIT-DOID and SNOMED-FMA (Body) datasets, along with tailored evaluation
metrics.
– Simplified file structure and task settings : We also re-organised the structure of the dataset
ifles, and reduced the task settings such that the unsupervised and semi-supervised settings
share the same testing set for ranking evaluation.</p>
        <p>For evaluation, in [33] we proposed both global matching and local ranking; the former aims
to evaluate the overall performance by computing Precision, Recall, and F1 metrics for the
output mappings against the reference mappings, while the latter aims to evaluate the ability to
distinguish the correct mapping out of several challenging negatives by ranking metrics Hits@K
and MRR. Note that subsumption mappings are inherently incomplete, so only local ranking
evaluation is applied for subsumption matching. For the special sub-track Bio-LLM introduced in
the 2023 edition, both matching and ranking metrics are used but they are tailored to the subsets,
along with an additional metric called rejection rate to examine if systems can reject all plausible
mappings for entities that actually have no alignment.
Ontology Pair
OMIM-ORDO</p>
        <p>NCIT-DOID
SNOMED-FMA
SNOMED-NCIT
SNOMED-NCIT</p>
        <p>Category
Disease
Disease</p>
        <p>Body
Pharm
Neoplas</p>
        <p>We adopted a flexible way of evaluating participating systems. First, participants can freely
choose any tasks and settings they would like to attend. Second, for systems that have been
well-adapted to the MELT platform, we used MELT to produce the output mappings. Third,
for systems that have been implemented elsewhere and are not easy to be made compatible
with MELT, we used their source code. Fourth, we also allowed participants (with trust) to
directly upload output mappings if their systems had not been published and had not been made
compatible with MELT. In the final result tables, we used superscripts †, ‡, and * to indicate
16Created from OMIM texts by Mondo’s pipeline tool avaiable at: https://github.com/monarch-initiative/omim.
17Created by the official snomed-owl-toolkit available at: https://github.com/IHTSDO/snomed-owl-toolkit.
that the results came from MELT, source code implementation, and direct result submission,
respectively. All our evaluations were conducted with the DeepOnto18 [38] library on a local
machine with Intel Xeon Bronze 3204 CPU 1.90GHz x11 processors, 126GB RAM, and two
Quadro RTX 8000 GPUs. The GPUs were mainly used for training systems that involve deep
neural networks.</p>
      </sec>
      <sec id="sec-3-5">
        <title>3.8. Biodiversity and Ecology</title>
        <p>The biodiversity and ecology (biodiv) track is motivated by the GFBio19 (The German Federation
for Biological Data) alongside its successor NFDI4Biodiversity20 and the AquaDiva21 projects,
which aim at providing semantically enriched data management solutions for data capture,
annotation, indexing and search [39, 40, 41]. In this track, we aim to motivate ontology matching
systems to work on matching ontologies and thesauri used in the biodiversity and ecology
domains, available via the BiodivPortal ontology repository22. For the current edition, we kept
the matching task between the Environment Ontology (ENVO) and the Semantic Web for Earth
and Environment Technology Ontology (SWEET) as these two ontologies have frequent updates.</p>
        <p>In 2021, we added a task to align two biological taxonomies with rather different but
complementary scopes: the well-known NCBI taxonomy (NCBITAXON), and TAXREF-LD [42]. No
matching system was able to achieve this matching task due to the large size of the considered
taxonomies. To cope with this issue since last year edition, we split the large matching task into a
set of smaller, more manageable subtasks through the use of modularization [43]. We obtained
six groups corresponding to the kingdoms: Animalia, Bacteria, Chromista, Fungi, Plantae and
Protozoa, leading to six well balanced matching subtasks. In 2023, we partnered with the
EcoPortal project23 to include two new matching tasks involving important thesauri in environmental
sciences (originally developed in SKOS): finding alignments between the Macroalgae Traits
Thesaurus (MACROALGAE) and the Macrozoobenthos Traits Thesaurus
(MACROZOOBENTHOS) and between the Fish Traits Thesaurus (FISH) and the Zooplankton Traits Thesaurus
(ZOOPLANKTON). Table 4 presents detailed information about the ontologies and thesauri used
in this year’s edition.</p>
      </sec>
      <sec id="sec-3-6">
        <title>3.9. Material Sciences and Engineering (MSE)</title>
        <p>Data in Material Sciences and Engineering (MSE) can be characterized by scarcity, complexity,
and the presence of gaps. Therefore the MSE community aims for ontology-based data integration
via decentralized data management architectures. Several actors using different ontologies result
in the growing demand for automatic alignment of ontologies in the MSE domain.</p>
        <p>The MSE track uses small to mid-sized ontologies common in the MSE field that are
implemented with and without upper-level ontologies. The ontologies follow heterogeneous design
18https://krr-oxford.github.io/DeepOnto/#/
19www.gfbio.org
20www.nfdi4biodiversity.org/en/
21www.aquadiva.uni-jena.de
22biodivportal.gfbio.org/
23ecoportal.lifewatch.eu/
principles with only partial overlap with each other. The current version v1.124 of the MSE
track includes three test cases summarized in Table 5, where each test case consists of two MSE
ontologies to be matched [ O1; O2] as well as one manual reference alignment R that can be
used for evaluation of the matching task. The benchmark also provides background knowledge
resources.</p>
        <p>The MSE track makes use of three different MSE ontologies in total, in each of which
an ontology using an upper-level ontology is matched to one without an upper-level. The
MaterialInformation[44] domain ontology was designed without upper-level ontology and serves
as infrastructure for material information and knowledge exchange (545 classes, 98 properties,
and 411 individuals). Three out of eight submodules of the MaterialInformation were merged
to create the Reduced MaterialInformation (32 classes, 43 properties, and 17 individuals) for
more efficient creation of the manual reference alignment in the First Test Case, see Table 5.
24https://github.com/EngyNasr/MSE-Benchmark/releases/tag/v1.1
The MatOnto Ontology v2.125 (847 classes, 96 properties and 131 individuals) bases on the
upper-level ontology bfo226. The Elementary Multiperspective Material Ontology (EMMO
v1.0.0-alpha2)27, is a standard representational ontology framework based on current materials
modeling and characterization knowledge incorporating an upper-, mid- and domain-level (451
classes, 35 properties). For every test case, a manual reference alignment R was created in close
cooperation with MSE domain experts.</p>
        <p>The evaluation was performed using the MELT platform on a Windows 11 system with Intel
Core i7-1260P CPU @2.10GHz and 16 GB RAM. Every participating system was executed in its
standard setting, and we compared precision, recall, and F-measure as well as the computation
time. No background knowledge was used for evaluation.</p>
      </sec>
      <sec id="sec-3-7">
        <title>3.10. Crosswalks Data Schema Matching</title>
        <p>This track was introduced in 2022, aiming at evaluating the ability of systems to deal with
the schema metadata matching task, in particular, with a collection of crosswalks from fifteen
research data schemes to Schema.org [45, 46]. It is based on the work carried out by the Research
Data Alliance (RDA) Research Metadata Schemas Working Group. The collection serves as a
reference for data repositories when they develop their crosswalks, as well as an indication of
semantic interoperability among the schemas.</p>
        <p>The dataset is composed of 15 source research metadata describing datasets that have been
aligned to Schema.org. The source schemas include discipline agnostic schemas Dublin Core,
Data Catalogue Vocabulary (DCAT), Data Catalogue Vocabulary - Application Profile
(DCATAP), Registry Interchange Format - Collections and Services (RIF-CS), DataCite Schema,
Dataverse; and discipline schemas ISO19115-1, EOSC/EDMI, Data Tag Suite (DATS), Bioschemas,
B2FIND, Data Documentation Initiative (DDI), European Clinical Research Infrastructure
Network (ECRIN), Space Physics Archive Search and Extract (SPASE); as well as CodeMeta for
software.</p>
        <p>This year a subset of the 15 metadata schemas aligned to schema.org has been considered. This
subset corresponds to the set of schemas and vocabularies for which an OWL/RDFS serialization
is available. It involves Data Catalogue Vocabulary (DCAT-v3), Data Catalogue Vocabulary
Application Profile (DCAT-AP), DataCity, Dublin Core (DC), and ISO19115-1 schemas (ISO).</p>
        <p>Using as a reference the manually established correspondences, the evaluation here will be
based on the well-known measures of precision, recall, and F-measure. The systems have been
executed on a Ubuntu Linux machine configured with 32GB of RAM running under an Intel Core
CPU 2.00GHz x8 processors.</p>
      </sec>
      <sec id="sec-3-8">
        <title>3.11. Common Knowledge Graphs</title>
        <p>This track was introduced to OAEI in 2021, and it evaluates the ability of matching systems
to match the schema (classes) in large cross-domain knowledge graphs such as DBpedia [47],
YAGO [48] and NELL [49]. The dataset used for the evaluation is generated from DBpedia and
25https://raw.githubusercontent.com/iNovexIrad/MatOnto-Ontologies/master/matonto-release.ttl
26http://purl.obolibrary.org/obo/bfo/2.0/bfo.owl
27https://raw.githubusercontent.com/emmo-repo/EMMO/1.0.0-alpha2/emmo.owl
the Never-Ending Language Learner (NELL). While DBpedia is generated from structured data in
Wikipedia’s articles, NELL is an automatically generated knowledge graph with entities extracted
from large-scale text corpus shared on websites. The automatic extraction process is one of the
aspects that make common knowledge graphs different from ontologies, as they often result in
less well-formatted and cross-domain datasets. In addition to the NELL and DBpedia test case,
this year, we introduced a new test case for matching classes from YAGO and Wikidata [50]. The
numbers of entities in the four KG datasets are illustrated in Table 6.</p>
        <p>The NELL and DBpedia benchmark [51] was human-annotated and verified by experts. This
gold standard is only a partial gold standard since not every class in each knowledge graph has
an equivalent class in the opposite one. To avoid over-penalizing matches that may discover
reasonable matches that are not included in the partial gold standard, our evaluation ignores
any predicted matches where neither of the classes in that pair exists in a true positive pair with
another class in the reference alignments. In terms of YAGO and Wikidata gold standard, it was
originally created [52] and expanded according to OAEI standard as part of [50].</p>
        <p>With respect to the reference alignment, matching systems were evaluated using standard
precision, recall, and f-measure. The evaluation was carried out on a Linux virtual machine
with 128 GB of RAM and 16 vCPUs (2.4 GHz) processors. The evaluation was performed
using MELT for matchers wrapped using both SEALS, and the web packaging via Docker. As a
baseline, we utilize a simple string matcher which is available through MELT.</p>
      </sec>
      <sec id="sec-3-9">
        <title>3.12. Knowledge Graph</title>
        <p>The Knowledge Graph track was run for the fourth year. The task of the track is to match pairs
of knowledge graphs whose schema and instances have to be matched simultaneously. The
individual knowledge graphs are created by running the DBpedia extraction framework on eight
different Wikis from the Fandom Wiki hosting platform28 in the course of the DBkWik project
[53, 54]. They cover different topics (movies, games, comics, and books) and three Knowledge
Graph clusters sharing the same domain e.g., star trek, as shown in Table 7.</p>
        <p>The evaluation is based on reference correspondences at both schema and instance levels.
While the schema-level correspondences were created by experts, the instance correspondences
were extracted from the wiki page itself. Due to the fact that not all interwiki links on a page
represent the same concept, a few restrictions were made: 1) only links in sections with a header
containing “link” are used, 2) all links are removed where the source page links to more than one
concept in another wiki (ensures the alignments are functional), 3) multiple links which point to
the same concept are also removed (ensures injectivity), 4) links to disambiguation pages were
manually checked and corrected. Since we do not have a correspondence for each instance, class,
and property in the graphs, this gold standard is only a partial gold standard.</p>
        <p>The evaluation was executed on a virtual machine (VM) with 32GB of RAM and 16 vCPUs
(2.4 GHz), with Debian 9 operating system and Openjdk version 1.8.0 265. For evaluating all
possible submission formats, MELT framework is used. The corresponding code for evaluation
can be found on Github29.</p>
        <p>The alignments were evaluated based on precision, recall, and f-measure for classes, properties,
and instances (each in isolation). The partial gold standard contained 1:1 correspondences, and
we further assume that in each knowledge graph, only one representation of the concept exists.
This means that if we have a correspondence in our gold standard, we count a correspondence to
a different concept as a false positive. The count of false negatives is only increased if we have a
1:1 correspondence and it is not found by a matcher.</p>
        <p>As a baseline, we employed two simple string-matching approaches. The source code for these
matchers is publicly available30.</p>
      </sec>
      <sec id="sec-3-10">
        <title>3.13. SPIMBENCH and Link Discovery</title>
        <p>This year, only LogMap has participated in the SPIMBENCH and Link Discovery tracks. The
organizers then made the decision not to run these tracks this year.</p>
      </sec>
      <sec id="sec-3-11">
        <title>3.14. Pharmacogenomics</title>
        <p>The Pharmacogenomics track is a new track proposed for OAEI 2023 that focuses on matching
knowledge units from the pharmacogenomics domain. These units are -ary tuples – so-called
“pharmacogenomic relationships” – and involve drugs, genetic factors, and phenotypes (see
Figure 1). A pharmacogenomic tuple states that patients being treated by the specified drugs
while having the specified genetic factors may experience the given phenotypes.</p>
        <p>In the Semantic Web formalisms, only binary predicates exist. That is why pharmacogenomic
tuples are reified: tuples become individuals that are linked to their components with binary
29https://github.com/dwslab/melt/tree/master/examples/kgEvalCli
30http://oaei.ontologymatching.org/2019/results/knowledgegraph/kgBaselineMatchers.zip
predicates (Figure 1(c)). Hence, the task of matching pharmacogenomic tuples is [55]:
– An instance matching task that aims at finding alignments between individuals representing
reified tuples;
– A structure-based matching task in which neighbors of reified tuples are compared to
conclude the potential alignment between tuples. Recall that no other information exist
about the tuples except their neighbors (e.g., no labels).</p>
        <p>To illustrate, two tuples involving the same drugs, genetic factors, and phenotypes will thus be
connected to the same neighbors and are expected to be detected as identical.</p>
        <p>Beside the arity of tuples, matchers need to face their potential incompleteness (e.g., missing
drugs) and heterogeneity (e.g., a gene version like CYP2C9*4 is more specific than the gene
itself CYP2C9, the phenotype hemorrhage is more specific than the phenotype vascular
disorders). Different types of alignments are thus expected to be identified between
pharmacogenomic tuples, which is somehow unusual in an instance matching task. The
Pharmacogenomics track features the identification of identical tuples ( =), equivalent tuples (Close), tuples
being more specific ( &lt;) or more general (&gt;) than others, and tuples being related to some extent
(Related). See [55, 56] for a detailed definition of these different alignment types between
individuals.</p>
        <p>To perform this alignment task, matchers can rely on additional background knowledge about
components of pharmacogenomic tuples. This knowledge includes ontology classes instanciated
by components of tuples (i.e. drugs, genetic factors, phenotypes) and their hierarchical
organization, partOf links between gene versions and genes, sameAs links between identical drugs,
genes, or phenotypes, and dependsOn links between complex phenotypes and their components
(e.g., “warfarin-induced bleeding” depends on “warfarin” and on “bleeding”).</p>
        <p>To evaluate matchers and their scalability, the Pharmacogenomics track comprises three tasks
involving respectively 10, 50, and 100% of the 50,435 pharmacogenomic tuples represented
within the PGxLOD knowledge graph31 [57]. For each task, the selected pharmacogenomic
tuples are evenly split into two ontologies to match. To take into account the specificity of the
different alignment types that are expected, matchers are evaluated through two settings:
Fine-grained setting Only alignments of the exact type expected in the reference are
considered correct. To illustrate, an output alignment (1, =, 2) where (1, Close, 2) was
expected will be considered as incorrect. Precision, Recall, and F1-score are computed for
each type of alignment.</p>
        <p>Coarse-grained setting Any type of alignment between entities expected to be aligned will be
considered as correct. To illustrate, an output alignment (1, =, 2) where (1, Close, 2)
was expected will be considered as correct. Precision, Recall, and F1-score are computed
globally accordingly.
{gf1, . . . , gf}</p>
        <p>{Codeine 25mg oral}
{p1, . . . , p}
{No effect}
warfarin causes</p>
        <p>causes pgt1 causes bleeding
{CYP2D6*4}</p>
        <p>CYP2C9*4
(a) Abstract relationship
(b) Example relationship
(c) Reified relationship</p>
      </sec>
    </sec>
    <sec id="sec-4">
      <title>4. Results and Discussion</title>
      <sec id="sec-4-1">
        <title>4.1. Participation</title>
        <p>Following an initial period of growth, the number of OAEI participants has remained
approximately constant since 2012, at slightly over 20. This year we count with 16 participating systems.
Table 8 lists the participants and the tracks in which they competed. It is worth mentioning that
the Bio-ML track has additional participants (e.g., BERTMap [58] and BERTSubs [59]) that are
not listed in Table 8. This is because they need training and validation which are not yet fully
supported by the OAEI evaluation platforms, and thus they were tested locally with Bio-ML
results reported, but without an OAEI system submission. Some matching systems participated
with different variants (Matcha and LogMap), whereas others were evaluated with different
configurations, as requested by developers (see test case sections for details). The following
sections summarize the results for each track.</p>
      </sec>
      <sec id="sec-4-2">
        <title>4.2. Anatomy</title>
        <p>The results for the Anatomy track are shown in Table 9. Of the 9 systems participating in the
Anatomy track, 8 achieved an F-measure higher than the StringEquiv baseline. Two systems were
ifrst-time participants (SORBETMatcher, OLaLa). Long-term participating systems showed few
changes in comparison with previous years with respect to alignment quality (precision, recall,
F-measure, and recall+), size and run time. The exception were LogMapBio which increase in
F-measure (from 0.895 to 0.898) and Matcha increased in recall+ (from 0.817 to 0.818), in size
(from 1482 to 1484). In terms of run time, 5 out of 9 systems computed an alignment in less
than 100 seconds. LogMapLt remains the system with the shortest runtime. Regarding quality,
Matcha achieved the highest F-measure (0.941) and recall+ (0.818), but four other systems
obtained an F-measure above 0.88 (OLaLa, SORBETMatcher, LogMapBio, and LogMap) which
is at least as good as the best systems in OAEI 2007-2010. Like in previous years, there is no
significant correlation between the quality of the generated alignment and the run time. Two
systems produced coherent alignments (LogMapBio and LogMap).</p>
      </sec>
      <sec id="sec-4-3">
        <title>4.3. Conference</title>
        <p>The conference evaluation results using the sharp reference alignment rar2 are shown in Table 10.
For the sake of brevity, only results with this reference alignment and considering both classes
and properties are shown. For more detailed evaluation results, please check the conference
track’s web page.</p>
        <p>With regard to two baselines we can group tools according to system’s position: six systems
outperformed above both baselines (GraphMatcher, SORBETMatcher, LogMap, Matcha, ALIN,
and OLaLa); three systems performed better than StringEquiv baseline (LogMapLt, AMD, and
LSMatch), and two systems performed worse than both baselines (TOMATO, and PropMatch).
Four matchers (AMD, ALIN, LSMatch, and SORBETMatcher) do not match properties at all. On
the other side, PropMatch does not match classes at all, while it dominates in matching properties.
Naturally, this has a negative effect on their overall performance.</p>
        <p>Several systems use reference alignments to a certain extent (GraphMatcher uses it in its 5-fold
cross-validation and TOMATO uses it in its sampling process) for their model learning, as has
happened with some ML-based systems in the past. These systems describe their usage in their
system papers.</p>
        <p>The performance of all matching systems regarding their precision, recall and F1-measure is
plotted in Figure 2. Systems are represented as squares or triangles, whereas the baselines are
represented as circles.</p>
        <p>The Conference evaluation results using the uncertain reference alignments are presented
in Table 11. Out of the 11 alignment systems, 6 (ALIN, AMD, LogMapLt, LSMatch,
SORBETMatcher, and TOMATO) use 1.0 as the confidence value for all matches they identify. The
remaining 5 systems (GraphMatcher, LogMap, Matcha, OLaLa, and PropMatch) have a wide
variation of confidence values.</p>
        <p>When comparing the performance of the systems on uncertain reference alignments versus the
sharp versions, it is evident that in the discrete cases, all systems either performed at the same
AMD
GraphMatcher
SORBETMatcher
LogMap
LogMapLt
LSMatch
Matcha
OLaLa
TOMATO
edna</p>
        <p>StringEquiv
level as before or showed improvements in terms of F-measure. The changes in F-measure for
discrete cases ranged from 1 to 16 percent above those in the sharp reference alignment. Notably,
LSMatch exhibited the most significant performance surge at (16%), closely followed by ALIN
at (15%) and AMD at (14%). This substantial improvement was primarily driven by increased
recall, a result of having fewer ’controversial’ matches in the uncertain version of the reference
alignment.</p>
        <p>In contrast, systems with confidence values consistently set at 1.0 delivered very similar
performance, regardless of whether a discrete or continuous evaluation methodology was applied.
This was due to their proficiency in identifying matches with which experts had a high degree of
agreement, while the matches they missed were typically the more contentious ones. Notably,
GraphMatcher stood out by producing the highest F-measure under both continuous (73%) and
discrete (73%) evaluation methodologies. This indicates that the system’s confidence evaluation
effectively reflects the consensus among experts in this task. However, it’s worth noting that
GraphMatcher experienced relatively small drops in F-measure when transitioning from discrete
to continuous evaluation, primarily due to a decrease in precision.</p>
        <p>In addition to the above findings, eight systems that participated this year were also part of the
previous year’s evaluation, allowing for some valuable comparisons over time. Among these, six
systems demonstrated remarkable stability in their F-measures when assessed against uncertain
reference alignments. However, two systems, Matcha and TOMATO, exhibited significant
improvements this year. Matcha’s F-measure jumped from (12%) in continuous and (14%) in
discrete last year to (63%) in continuous and (65%) in discrete this year, primarily due to an
increase in precision. Similarly, TOMATO saw a substantial increase in F-measure from (15%)
last year to (62%) this year, both in continuous and discrete evaluations. These improvements
mark significant progress for these two systems compared to the previous year.</p>
        <p>OLaLa, PropMatch, and SORBETMatcher are three new systems participating this year.
OLaLa has shown notable performance improvements, with a (4%) increase in the discrete case
and a (2%) increase in the continuous case concerning F-measure when compared to the sharp
reference alignment. OLaLa’s F-measure has risen from (61%) to (65%) in the discrete case and
to (63%) in the continuous case. This improvement is primarily attributed to an increase in recall.</p>
        <p>On the other hand, PropMatch and SORBETMatcher have demonstrated similar performance
in both discrete and continuous cases when compared to the sharp reference alignment in terms
of F-measure. Notably, PropMatch exhibits consistently lower precision and recall across the
three different versions of the reference alignment, primarily due to its narrow focus on property
matching.</p>
      </sec>
      <sec id="sec-4-4">
        <title>4.4. Multifarm</title>
        <p>This year, 4 systems have registered to participate in the Multifarm track:LSMatch-Multilingual,
LogMap, LogMapLt, and Matcha. The number of participating tools is stable with respect to the
last 4 campaigns (5 in 2022, 6 in 2021, 6 in 2020, 5 in 2019, 6 in 2018, 8 in 2017, 7 in 2016, 5 in
2015, 3 in 2014, 7 in 2013, and 7 in 2012). This year, we lost the participation of CIDER-LM.
The reader can refer to the OAEI papers for a detailed description of the strategies adopted by
each system.</p>
        <p>The Multifarm evaluation results based on the blind dataset are presented in Table 12,
demonstrating the aggregated results for the matching tasks. They have been computed using the MELT
framework without applying any threshold to the results. They are measured in terms of macro
precision and recall. The results of non-specific systems are not reported here, as we could
observe in the last campaigns that they can have intermediate results in tests of type ii) (same
ontologies task) and poor performance in tests i) (different ontologies task).</p>
        <p>The systems have been executed on a Ubuntu Linux machine configured with 32GB of RAM
LogMapLt stands out for its very fast calculation time of 5s (resp. 6s) to find “equal” (resp.
“subclass” relation correspondences). Concerning “equal” (=) relation correspondences, LogMap
and LogMapLt have better precision than Matcha and OLaLa. However, LogMap’s recall is
20 (resp. 11 times) less than OLaLa’s (resp. Matcha’s) one. Matcha is the best-performing
participant in the FNC test case in terms of precision and F1-measure. None of the matching
systems are able to find “subclass” relation ( &lt;) correspondences.</p>
      </sec>
      <sec id="sec-4-5">
        <title>4.7. Interactive matching</title>
        <p>This year, two systems (ALIN, and LogMap) participated in the Interactive matching track. Their
results are shown in Table 14 and Figure 3 for both the Anatomy and Conference datasets.</p>
        <p>The table includes the following information (column names within parentheses):
– The performance of the system: Precision (Prec.), Recall (Rec.), and F-measure (F-m.)
with respect to the fixed reference alignment, as well as Recall+ (Rec.+) for the Anatomy
task. To facilitate the assessment of the impact of user interactions, we also provide the
performance results from the original tracks, without interaction (line with Error NI).
– To ascertain the impact of the oracle errors, we provide the performance of the system with
respect to the oracle (i.e., the reference alignment as modified by the errors introduced
by the oracle: Precision oracle (Prec. oracle), Recall oracle (Rec. oracle) and F-measure
oracle (F-m. oracle). For a perfect oracle, these values match the actual performance of the
system.
– Total requests (Tot Reqs.) represents the number of distinct user interactions with the tool,
where each interaction can contain one to three conflicting correspondences, that could be
analyzed simultaneously by a user.
– Distinct correspondences (Dist. Mapps) counts the total number of correspondences for
which the oracle gave feedback to the user (regardless of whether they were submitted
simultaneously, or separately).
NI stands for non-interactive, and refers to the results obtained by the matching system in the
original track.
– Finally, the performance of the oracle itself with respect to the errors it introduced can be
gauged through the positive precision (Pos. Prec.) and negative precision (Neg. Prec.),
which measure respectively the fraction of positive and negative answers given by the
oracle that are correct. For a perfect oracle, these values are equal to 1 (or 0, if no questions
were asked).</p>
        <p>The figure shows the time intervals between the questions to the user/oracle for the different
systems and error rates. Different runs are depicted with different colors.</p>
        <p>The matching systems that participated in this track employ different user-interaction strategies.
While LogMap makes use of user interactions exclusively in the post-matching steps to filter
their candidate correspondences, ALIN can also add new candidate correspondences to its initial
set. LogMap requests feedback on only selected correspondences candidates (based on their
similarity patterns or their involvement in unsatisfiabilities). ALIN and LogMap can both ask the
oracle to analyze several conflicting correspondences simultaneously.</p>
        <p>The performance of the systems usually improves when interacting with a perfect oracle in
comparison with no interaction. ALIN is the system that improves the most, because of its high
number of oracle requests, and its non-interactive performance was the lowest of the interactive
systems, and thus the easiest to improve.</p>
        <p>Although system performance deteriorates when the error rate increases, there are still benefits
from the user interaction—some of the systems’ measures stay above their non-interactive values
even for the larger error rates. Naturally, the more a system relies on the oracle, the more its
performance tends to be affected by the oracle’s errors.</p>
        <p>The impact of the oracle’s errors is linear for ALIN in most tasks, as the F-measure according
to the oracle remains approximately constant across all error rates. It is supra-linear for LogMap
in all datasets.</p>
        <p>Another aspect that was assessed, was the response time of systems, i.e., the time between
requests. Two models for system response times are frequently used in the literature [60]:
Shneiderman and Seow take different approaches to categorize the response times taking a
taskcentered view and a user-centered view respectively. According to task complexity, Shneiderman
defines response time in four categories: typing, mouse movement (50-150 ms), simple frequent
tasks (1 s), common tasks (2-4 s) and complex tasks (8-12 s). While Seow’s definition of response
time is based on the user expectations towards the execution of a task: instantaneous (100-200
ms), immediate (0.5-1 s), continuous (2-5 s), captive (7-10 s). Ontology alignment is a cognitively
demanding task and can fall into the third or fourth categories in both models. In this regard
the response times (request intervals as we call them above) observed in all datasets fall into the
tolerable and acceptable response times, and even into the first categories, in both models. The
request intervals for LogMap and ALIN stay at a few milliseconds for most datasets. It could be
the case, however, that a user would not be able to take advantage of these low response times
because the task complexity may result in higher user response time (i.e., the time the user needs
to respond to the system after the system is ready).
4.8. Bio-ML
Our results include vfie tables for equivalence matching, vfie tables for subsumption matching,
and two tables for Bio-LLM, where each table corresponds to an OM pair and includes results of
both the unsupervised and semi-supervised settings. For the full results, please refer to the OAEI
2023 Bio-ML website32.</p>
        <p>Briefly, we have the following participants for equivalence matching: (i) machine
learningbased systems including BERTMap, BERTMapLt [58], AMD [61], Matcha, Matcha-DL [62],
OLala, and SORBETMatcher [63]; and (ii) traditional systems including LogMap, LogMapBio,
and LogMapLt [36]. For equivalence matching, top performing systems vary across different tasks,
with LogMapBio attaining the best F1 on 2 out of 5 unsupervised tasks, and AMD, BERTMap, and
SORBETMatcher attaining the best F1 on each of the remaining three, respectively. BERTMap
also attains the best ranking scores of all tasks, though most systems do not provide ranking
results in equivalence matching. For subsumption matching, all the participating systems are
machine learning-based, including BERTSubs (IC) [59], OWL2Vec*+RF [64], SORBETMatcher
[63], and Word2Vec+Random Forest (RF). SORBETMatcher attains the best MRR on 3 out of 5
subsumption tasks, while BERTSubs (IC) and OWL2Vec*+RF attain the best MRR on each of
32https://www.cs.ox.ac.uk/isg/projects/ConCur/oaei/2023/, permanent link at Internet Wayback Machine, https://web.</p>
        <p>archive.org/web/20231120093331/https://www.cs.ox.ac.uk/isg/projects/ConCur/oaei/2023/
the remaining two, respectively. Overall, the 2023 edition attracted more machine learning-based
participants, which matches the original purpose of Bio-ML, while LogMap variants are the only
symbolic participants.</p>
      </sec>
      <sec id="sec-4-6">
        <title>4.9. Biodiversity and Ecology</title>
        <p>This year, vfie matching systems (LogMap, LogMapLt, LogMapKG, Matcha and OLaLa)
managed to generate an output for all of the track tasks, except Matcha failed to achieve alignment for
the envo-sweet task. As in previous editions, we used precision, recall, and F-measure to evaluate
the performance of the participating systems. The results for the Biodiversity and Ecology track
are shown in Table 15.</p>
        <p>In comparison to the previous year, a smaller number of systems succeeded in generating
alignments for the track tasks. The results of the participating systems are comparable to last year in
terms of F-measure. In terms of run time, OLaLa took the longer. Regarding the ENVO-SWEET
task, only OLaLa and the LogMap family systems achieved it with a similar performance to last
year. The MACROALGAE-MACROZOOBENTHOS and FISH-ZOOPLANKTON matching
tasks involve resources developed in SKOS. For the transformation, we made use of a source
code directly derived from the AML ontology parsing module, kindly provided to us by its
developers. The systems that did not perform well in this task did map a large number of dissimilar
concepts that happen to have similar URIs. All systems performed well on most
NCBITAXONTAXREF-LD subtasks, with slightly the same levels of precision and recall. Overall, in this year’s
evaluation, the number of participating systems decreased and the performance of the successful
ones remained similar.</p>
      </sec>
      <sec id="sec-4-7">
        <title>4.10. Material Sciences and Engineering (MSE)</title>
        <p>This year four systems registered on the MSE track, each of which was used for evaluation with
the three test cases of the MSE benchmark. AMD produced errors and an empty alignment file, so
results are only available for three of the matchers: LogMap, LogMapLt, Matcha. The evaluation
results are shown in Table 16.</p>
        <p>The first test case evaluates matching systems regarding their capability to find “equal” (=),
“superclass” (&gt;) and “subclass” (&lt;) correspondences between the mid-sized MatOnto and the
small-sized (since reduced) MaterialInformation ontology. None of the evaluated systems finds
correspondences other than “equal” (=). All evaluated systems compute the alignment in less
than a minute. In contrast to the results of 2022 Matcha performs the matching task almost as
quickly as LogMap, which stood out for its very fast calculation time. Apart from the runtime,
the results for LogMap and LogMapLt do not change in comparison to the evaluation in 2022.
LogMap presents a maximum precision value of 1.0, however since only one correspondence
was found by LogMap, the recall and hence the F1-measure is low (0.083). In direct comparison
to LogMap, LogMapLt calculates the alignment in around half the time and achieves much
lower precision (0.4) but due to a greater amount of correctly found correspondences, the
F1-measure is better - although still low with 0.143. Matcha finds 8 incorrect correspondences
and thus is the worst-performing participant in terms of precision in the first test case.
Investigating the reason for this low precision, Matcha appears to match classes with object</p>
        <p>Time Number of Precision Recall F-measure
(HH:MM:SS) mappings
properties, e.g. “Temperature” = “hasTemperature” as in 2022. Since the recall is the best of the
participating systems, Matcha turns out to be the best-performing system based on its F1-measure.</p>
        <p>The second test case evaluates the matching systems to find correspondences between the
large-sized MaterialInformation and the mid-sized BFO-based MatOnto. Surprisingly, LogMap
performs the matching task significantly quicker than in the first test case and stands out for its
very fast computation time of only 6s at a high precision of 0.881. Apart from the runtime, the
results for LogMap and LogMapLt do not change in comparison to the evaluation in 2022. Since
LogMap found only 59 correct correspondences out of the 302 reference correspondences, the
recall is rather low, but the F1-measure is still the highest of the tested systems. LogMapLt
is significantly slower than LogMap but finds the same amount of correspondences with 2
additional false positives, so it achieves a slightly lower overall F1-measure than LogMap.
Matcha finds 6 wrong correspondences where classes are matched to object properties as in the
ifrst test case. Matcha presents the lowest precision in this test case but the highest recall. Based
on its F1-measure, Matcha performs slightly better than the other systems in this test case.</p>
        <p>The third test case evaluates matching systems to find correspondences between the large-sized
MaterialInformation and the mid-sized EMMO. All evaluated systems compute the alignments
in under one minute. Apart from the runtime, the results for LogMap and LogMapLt do not
change in comparison to the evaluation in 2022. All of the systems present high precision values.
LogMap computes 3 false positives, LogMapLt computes 5 false positives and Matcha 3 false
positives. At the same time Matcha computes with 56 the highest number of true positives, which
results in the best precision of the tested systems. Since it has the highest number of correctly
found correspondences, the recall and the F1-measure are the highest of the evaluated systems in
the third test case at the fastest computation time.</p>
        <p>In summary, LogMap stands out for its very fast computing speed with very high precision at
the same time. LogMapLt is significantly slower in every test case and almost constantly shows
worse results - only in the first test case the recall of LogMapLt is higher than for LogMap. In
our opinion, LogMap is definitely recommended for MSE applications where high precision is
demanded. In comparison to that, LogMapLt does not appear to bring any decisive advantage
over LogMap.</p>
        <p>Matcha in its current implementation is not recommended for MSE applications since it
matches classes to properties.</p>
        <p>A-LIOn produces moderate results but does not bring any advantage over LogMap.
Furthermore, A-LIOn produces errors while reasoning on EMMO. The latter is the only one of the MSE
ontologies used with a significant proportion of essential axioms. According to the annotations in
EMMO, this ontology exclusively can be inferred with the FaCT++ reasoner. That might be a
cause for the occurring reasoning errors of A-LIOn and bad results in the third test case.</p>
        <p>None of the evaluated matcher finds all reference correspondences correctly and none of the
matchers.</p>
      </sec>
      <sec id="sec-4-8">
        <title>4.11. Common Knowledge Graphs</title>
        <p>We evaluated all the participating systems that were packaged as SEALS packages or as web
services using Docker (even those not registered to participate on this new track). However, not
all systems were able to complete the task, as some systems finished with an empty alignment
ifle. Here, we include the results of 7 systems that were able to finish the task within the 24-hour
time limit with a non-empty alignment file: LogMap, OLaLa, Matcha, LogMapLt, LogMapKG,
LsMatch, and AMD.</p>
        <p>Table 17 shows the aggregated results on the two datasets for systems that produced non-empty
alignment files. The size column indicates the total number of class alignments discovered
by each system. While the majority of the systems discovered alignments at both schema and
instance levels, we have only evaluated class alignments, as the two gold standard does not
include any instance-level ground truth. Further, Not all systems were able to handle the original
dataset versions (i.e., those with all annotated instances). In terms of the NELL-DBpedia test
case, LogMap, OLaLa, Matcha, and AMD were able to generate results when applied to the
full-size dataset. While on the YAGO-Wikidata dataset, which is large-scale compared to the
ifrst dataset, only OLaLa, Matcha, and AMD were able to generate alignments with the original
dataset. Other systems either fail to complete the task within the allocated 24-hour time limit
such as LogMapLt and LsMatch, or produce an empty alignment file such as LogMap (only on
the YagoWikidata dataset). LogMapKG on the other hand tends to only align instances when it is
applied to full-size datasets. Similar to the 2022 evaluation results, AMD does generate schema
alignments but in the wrong format, therefore, they can not be evaluated.</p>
        <p>The resulted alignment files from all the participating systems are available to download on the
track’s result webpage33. On the Nell-DBpedia dataset, all systems were able to outperform the
basic string matcher, in terms of f-measure, except for LogMapLt. On the YagoWikidata dataset,
two systems were not able to outperform the baseline, which are LogMapLt and LsMatch. This
year saw the return of different matchers and the introduction of a new one, OLaLa. While most
LogMap</p>
        <p>OLaLa
LogMapLt
LogMapKG</p>
        <p>AMD
LsMatch</p>
        <p>Matcha
String Baseline</p>
        <p>LogMap</p>
        <p>OLaLa
LogMapLt
LogMapKG</p>
        <p>AMD
LsMatch</p>
        <p>Matcha
String Baseline
105
120
77
104
102
101
114
78
233
209
211
232
125
196
274
212
matchers demonstrated similar performance to previous evaluations, Matcha notably improved its
results on both datasets. Matcha also showcased the ability to function with the original datasets,
a capability it lacked in the 2022 evaluation. OLaLa outperformed all other matchers in the
Nell-DBpedia task, whereas Matcha excelled on the larger dataset, Yago-Wikidata. Furthermore,
all matching processes were completed in less than an hour, as indicated in the runtime column.
Lastly, the dataset size column specifies whether a system operated on the original dataset or
solely on the smaller version.</p>
      </sec>
      <sec id="sec-4-9">
        <title>4.12. Crosswalks Data Schema Matching</title>
        <p>All the systems registered to OAEI 2023 were run besides the fact that only LogMap has
been registered to participate in all tracks and no system has been specifically registered to the
Crosswalks task.</p>
        <p>This year, as introduced above, we have used the schemes for which an OWL/RDFS
serialization is available, as OAEI matching systems are used to the format. However, this does not reflect
the reality of the field, as schemes are not usually exposed in such a structured format. This opens
the possibility of providing a dedicated task next year.</p>
        <p>Table 18 shows the results for the systems that have generated correspondences. While
generating a few number of correct correspondences, precision is higher with respect to recall for
datacity
iso
dcat3
dcterms
dcat-ap
datacity
iso
dcat3
dcterms
dcat-ap
datacity
iso
dcat
dcterms
dcat-ap
datacity
iso
dcat3
dcterms
dcat-ap
datacity
iso
dcat3
dcterms
dcat-ap
0
0
3 1
0
0
1
0
0
1
0
0
1
0
0
1
1
0
2
0
1
4
1
0
6
0
1
4
1
0
6
34
42
42
32
34
184
all systems. Most of the generated correspondences still involve properties where labels are equal,
for instance: https://schema.org/distribution and http://www.w3.org/ns/dcat#distribution. In terms
of F-measure, Matcha and LogMapLt have the best and similar performance. With respect to the
pairs, a higher number of correspondences has been generated for the pairs involving DCAT-v3.
LogMapLt and Matcha are the systems that are able to deal with a higher number of matching
pairs.</p>
        <p>In 2022, this track ran for the first time. Last year, similar to this year, only Matcha, LogMap
and LogMapLt were able to generate non-empty alignments, with LogMapLt being able to
generate a higher number of correspondences. In terms of precision, Matcha and LogMapLt had
a higher precision in detriment of recall.</p>
      </sec>
      <sec id="sec-4-10">
        <title>4.13. Knowledge Graph</title>
        <p>This year we evaluated all participants with the MELT framework to include all possible
submission formats i.e. SEALS, and Web format. First, all systems are evaluated on a very small
matching task34 (even those not registered for the track). This revealed that not all systems were
able to handle the task, and in the end, 6 matchers can provide results for at least one test case.</p>
        <p>Similar to the previous years, some systems like AMD need a post-processing step of the
resulting alignment file to be able to parse it. The reason is that the KGs in the knowledge graph
track contain special characters, e.g. ampersand. These characters need to be encoded in order to
parse these XML-formatted files correctly. The resulting alignments are available for download
35.</p>
        <p>Table 19 shows the results for all systems divided into class, property, instance, and overall
results. This also includes the number of tasks in which they were able to generate a non-empty
alignment (#tasks) and the average number of generated correspondences (size). We report the
macro averaged precision, F-measure, and recall results, where we do not distinguish empty
and erroneous (or not generated) alignments. The values in parentheses show the results when
considering only nonempty alignments.</p>
        <p>This year’s best overall system is the baseline using the alternative labels (0.84 F-measure).
The highest recall is again achieved by Matcha (0.84). It returns more correspondences than all
others (263,822.2 on average) but is only able to match instances in this track. Detailed results
for each test case can be found on the OAEI results page of the track36.</p>
        <p>Property matches are still not created by all systems. LogMap, Matcha, and
SORBETMatcher do not return any of those mappings. One reason might be that the
properties are typed as rdf:Property and not distinguished into owl:ObjectProperty or
owl:DatatypeProperty. OLaLa reaches the best score with 0.83 F-Measure.</p>
        <p>Regarding runtime, Matcha (14:30:03) and LogMapLt (64:48:07) were the slowest systems. In
comparison to last year, the runtimes increased quite a lot and the systems should focus more
on scalable solutions. Besides the baselines (which need around 12 minutes for all test cases)
LogMap (00:56:43) and SORBETMatcher (00:21:53) were the fastest systems.</p>
        <p>For further analysis of the results, we also provide an online dashboard37 generated with
MELT[65]. It allows us to inspect the results on a correspondence level. Due to the large amount
of these correspondences, it can take some time to load the full dashboard.</p>
      </sec>
      <sec id="sec-4-11">
        <title>4.14. Pharmacogenomics</title>
        <p>For this first year of the Pharmacogenomics track, 2 systems registered, namely LogMap and
Matcha. The evaluation was performed using the MELT framework. Unfortunately, none
34http://oaei.ontologymatching.org/2019/results/knowledgegraph/small test.zip
35http://oaei.ontologymatching.org/2023/results/knowledgegraph/knowledgegraph-alignments.zip
36http://oaei.ontologymatching.org/2023/results/knowledgegraph/index.html
37http://oaei.ontologymatching.org/2023/results/knowledgegraph/knowledge graph dashboard.html
BaselineAltLabel
BaselineLabel
LogMap
LogMapLt
LSMatch
Matcha
OLaLa
SORBETMatcher
BaselineAltLabel
BaselineLabel
LogMap
LogMapLt
LSMatch
Matcha
OLaLa
SORBETMatcher
BaselineAltLabel
BaselineLabel
LogMap
LogMapLt
LSMatch
Matcha
OLaLa
SORBETMatcher
BaselineAltLabel
BaselineLabel
LogMap
LogMapLt
LSMatch
Matcha
OLaLa
SORBETMatcher
of these systems successfully produced alignments between reified -ary tuples representing
pharmacogenomic knowledge units. Some configurations of the two systems output some
alignments between other entities (e.g., components of pharmacogenomic tuples) but not between
the -ary tuples themselves.</p>
      </sec>
    </sec>
    <sec id="sec-5">
      <title>5. Conclusions and Lessons Learned</title>
      <p>As in previous campaigns, in 2023, we witnessed a healthy mix of new and returning systems,
with an imbalanced participation in the tracks.</p>
      <p>The schema matching tracks gather the highest number of participants; however still little
substantial progress in terms of the quality of the results or run time of top matching systems. As
already reported last year, we observe a performance plateau being reached by existing strategies
and algorithms. It is also true that established matching systems tend to focus more on new tracks
and datasets than on improving their performance in long-standing tracks, whereas new systems
typically struggle to compete with established ones.</p>
      <p>According to the Conference track, there are more systems with the ability to match properties
(7 in 2023 vs. 5 in 2022). Several ML-based systems used reference alignments for training to a
certain extent (explained in their system papers). It has already happened in the past. It calls for a
discussion and perhaps more ML-based tracks.</p>
      <p>Since the creation of the Material Sciences and Engineering track, a large amount of new
ontologies have been developed and utilized in various MSE applications. In contrast to the early
development stages of this track, those ontologies are now easily accessible on the Matportal38.
In the future, the MSE track should be updated with the currently most used top and mid-level
MSE ontologies, which include the BWMD-mid39, the MSEO40, the PMDco41, the prov-o42 and
IOF-mat43. In the OntoCommons-project44 alignments of frequently used ontologies of the MSE
application area are produced and will be used to further improve the MSE benchmark based on
the project results. Apart from also considering frequently used domain and application ontologies,
also multi-ontology matching, knowledge graph matching, e.g. using the AluTrace-data45, and
the usage of background knowledge should be considered in future OAEI campaigns.</p>
      <p>With respect to the cross-lingual version of the Conference, the Multifarm track still attracts
too few number of participants. Despite this fact, this year new participants came up with
alternative strategies (i.e., deep learning) with respect to the last campaigns.</p>
      <p>In the Food track, none of the evaluated matchers finds all reference correspondences correctly.
LogMapLt stands out for its very fast computing speed. Matcha obtains the best results for
the FNC application. The usage of background knowledge available in CIQUAL and SIREN
ontologies in terms of food description based on FoodON concepts should be considered in future
OAEI campaigns.</p>
      <p>The Bio-ML track incorporated significant updates and attracted several new machine
learningbased participants. However, the number of symbolic participants decreased. The best-performing
systems are not consistent across tasks and settings, demonstrating the diversity of our datasets. It
is also worth noting that SORBETMatcher is the only system can participate in both equivalence
38https://matportal.org/
39https://matportal.org/ontologies/BWMD-MID
40https://matportal.org/ontologies/MSEO
41https://github.com/materialdigital/core-ontology/
42https://www.ebi.ac.uk/ols/ontologies/prov
43https://industrialontologies.org/working-groups/the-material-science-and-engineering-mse-working-group-wg/
44https://ontocommons.eu
45https://github.com/Mat-O-Lab/AluTraceProject
and subsumption matching.</p>
      <p>In the Biodiversity and Ecology track, none of the systems was able to detect manual mappings
created by domain experts and requiring biodiversity domain-specific knowledge. In this year’s
edition, we confirmed the inability of most systems to handle SKOS natively, as well as very
large ontologies. Additionally, some systems did not perform well on the thesauri tasks because
those contained concepts with similar URIs that were, in fact, completely different.</p>
      <p>The Interactive matching track also witnessed a small number of participants. Two systems
participated this year. This is puzzling considering that this track is based on the Anatomy and
Conference test cases, and those tracks had 9 and 11 participants, respectively. The process of
programmatically querying the Oracle class used to simulate user interactions is simple enough
that it should not be a deterrent for participation, but perhaps we should look at facilitating the
process further in future OAEI editions by providing implementation examples.</p>
      <p>The Complex matching track tackles a challenge task that attracts too few number of
participants. This year, no system was able to complete the task. As several sub-tracks have been
discontinued, the track is limited to the conference domain. This track welcomes new organizers.</p>
      <p>The Crosswalks Data Schema Matching track involves different schema formats and ways
of representing schema properties. This opens the possibility of creating a dedicated task relying
on other formats than OWL/RDF.</p>
      <p>Automatic instance-matching benchmark generation algorithms have been gaining popularity,
as evidenced by the fact that they are used in all three instance-matching tracks of this OAEI
edition. One aspect that has not been addressed in such algorithms is that, if the transformation is
too extreme, the correspondence may be unrealistic and impossible to detect even by humans. As
such, we argue that human-in-the-loop techniques can be exploited to do a preventive
qualitychecking of generated correspondences and refine the set of correspondences included in the final
reference alignment.</p>
      <p>In the Knowledge graph track, the overall best scores are still unbeaten. Furthermore, the
proportion of matchers not able to produce property alignments is high. This might change next
year with new and improved systems.</p>
      <p>In the Common knowledge graphs track, which challenges matching systems to map the
schema of large-scale, automatically constructed, and cross-domain knowledge graphs. The
number of participants is similar to last year, with a new system participating and former systems
adapting their approaches to scale up to the task size. However, with some systems only being
able to produce alignments when applied to smaller versions of the KG datasets, we still look
forward to having more participants in the next OAEI campaign.</p>
      <p>For the first year of the Pharmacogenomics track, participation was limited with only 2
systems registered. Unfortunately, none of the participating systems were able to output alignments
between the targeted reified -ary tuples. We will investigate whether this originates from the
absence of labels for tuples and the only presence of structural information or if other aspects
are detrimental (e.g., arity, background domain knowledge that must be considered to produce
most alignments). These results highlight the interest in considering domain-specific problems to
design new methods like [56, 66] or enrich existing ones. Recall that the track features different
types of alignments between individuals, which is a specificity of the considered alignment task.
This raises the question of whether such a granular matching setting could be generalized to other
instance matching tasks. Since the alignment task in this track is structure-based, it is particularly
well-adapted to approaches relying on Graph Neural Networks that learn embeddings of nodes to
align on the basis of their neighborhoods [66]. All these reasons motivate to propose again the
track in the next editions of OAEI and adapt it to evaluate Machine Learning-based matchers. We
hope that the growing awareness about this track and its specificity will attract additional systems.</p>
      <p>Like in previous OAEI editions, most participants provided a description of their systems and
their experience in the evaluation, in the form of OAEI system papers. These papers, like the
present one, have not been peer-reviewed. However, they are full contributions to this evaluation
exercise, reflecting the effort and insight of matching systems developers, and providing details
about those systems and the algorithms they implement.</p>
      <p>As each year, fruitful discussions at the Ontology Matching Workshop point out different
directions for future improvements in OAEI. This year, with a higher number of systems relying
on Large Language Models, there was a discussion on the specific requirements and alternative
ways for gathering the alignments generated by such resource-consuming systems. It has also
been highlighted the need to push the adoption of SSSOM [24] (this year MELT has incorporated
the format but still few systems have adopted it), as a way for delivering richer alignments in
terms of metadata and justifications [ 67]. As already mentioned before, there were also some
interrogations on the stability reached in some (open)-schema matching tasks (in particular
Anatomy and Conference tracks) as the performance has been quite stable for several years. This
requires a further analysis of the difcfiult parts of the matching task. Last but not least, new
tracks addressing more application/use-oriented tasks should be addressed and they are more than
welcome.</p>
      <p>The Ontology Alignment Evaluation Initiative will strive to remain a reference to the ontology
matching community by improving both the test cases and the testing methodology to better
reflect actual needs, as well as to promote progress in this field. More information can be found
at: http://oaei.ontologymatching.org.</p>
    </sec>
    <sec id="sec-6">
      <title>Acknowledgments</title>
      <p>We warmly thank the participants of this campaign. We know that they have worked hard to have
their matching tools executable in time and they provided useful reports on their experience. The
best way to learn about the results remains to read the papers that follow.</p>
      <p>We are also grateful to Martin Ringwald and Terry Hayamizu for providing the reference
alignment for the anatomy ontologies and thank Elena Beisswanger for her thorough support in
improving the dataset’s quality.</p>
      <p>We also thank for their support, the past members of the Ontology Alignment Evaluation
Initiative steering committee: Je´roˆme Euzenat (INRIA, FR), Yannis Kalfoglou (Ricoh laboratories,
UK), Miklos Nagy (The Open University, UK), Natasha Noy (Google Inc., USA), Yuzhong Qu
(Southeast University, CN), York Sure (Leibniz Gemeinschaft, DE), Jie Tang (Tsinghua University,
CN), Heiner Stuckenschmidt (Mannheim Universita¨t, DE), and George Vouros (University of the
Aegean, GR).</p>
      <p>Daniel Faria and Catia Pesquita were supported by the FCT through the LASIGE Research Unit
(UIDB/00408/2020 and UIDP/00408/2020) and by the KATY project funded by the European
Union’s Horizon 2020 research and innovation program under grant agreement No 101017453.</p>
      <p>Ernesto Jimenez-Ruiz has been partially supported by the SIRIUS Centre for Scalable Data
Access (Research Council of Norway, project no.: 237889).</p>
      <p>Irini Fundulaki and Tzanina Saveta were supported by the EU’s Horizon 2020 research and
innovation program under grant agreement No 688227 (Hobbit).</p>
      <p>Patrick Lambrix, Huanyu Li, Mina Abd Nikooie Pour and Ying Li have been supported by
the Swedish e-Science Research Centre (SeRC) and the Swedish National Graduate School in
Computer Science (CUGS).</p>
      <p>Beyza Yaman has been supported by ADAPT SFI Research Centre [grant 13/RC/2106 P2].</p>
      <p>Jiaoyan Chen, Hang Dong, Yuan He, and Ian Horrocks have been supported by Samsung
Research UK (SRUK) and the EPSRC project ConCur (EP/V050869/1).</p>
      <p>Naouel Karam and Alsayed Algergawy have been supported by the German Research
Foundation in the context of NFDI4BioDiversity project (number 442032008) and the CRC 1076
AquaDiva. We would like to thank Jessica Titocci, Martina Pulieri and Ilaria Rosati for providing
the datasets for the biodiv SKOS thesauri tasks.
2021, volume 3063 of CEUR Workshop Proceedings, CEUR-WS.org, 2021, pp. 62–108.</p>
      <p>URL: http://ceur-ws.org/Vol-3063/oaei21 paper0.pdf.
[7] M. Abd Nikooie Pour, A. Algergawy, R. Amini, D. Faria, I. Fundulaki, I. Harrow, S. Hertling,
E. Jime´nez-Ruiz, C. Jonquet, N. Karam, A. Khiat, A. Laadhar, P. Lambrix, H. Li, Y. Li,
P. Hitzler, H. Paulheim, C. Pesquita, T. Saveta, P. Shvaiko, A. Splendiani, E´. Thie´blin,
C. Trojahn, J. Vatascinova´, B. Yaman, O. Zamazal, L. Zhou, Results of the ontology
alignment evaluation initiative 2020, in: P. Shvaiko, J. Euzenat, E. Jime´nez-Ruiz, O.
Hassanzadeh, C. Trojahn (Eds.), Proceedings of the 15th International Workshop on Ontology
Matching co-located with the 19th International Semantic Web Conference (ISWC 2020),
Virtual conference (originally planned to be in Athens, Greece), November 2, 2020,
volume 2788 of CEUR Workshop Proceedings, CEUR-WS.org, 2020, pp. 92–138. URL:
http://ceur-ws.org/Vol-2788/oaei20 paper0.pdf.
[8] A. Algergawy, D. Faria, A. Ferrara, I. Fundulaki, I. Harrow, S. Hertling, E. Jime´nez-Ruiz,
N. Karam, A. Khiat, P. Lambrix, H. Li, S. Montanelli, H. Paulheim, C. Pesquita, T. Saveta,
P. Shvaiko, A. Splendiani, E´. Thie´blin, C. Trojahn, J. Vatascinova´, O. Zamazal, L. Zhou,
Results of the ontology alignment evaluation initiative 2019, in: Proceedings of the 14th
International Workshop on Ontology Matching, Auckland, New Zealand, 2019, pp. 46–85.
[9] A. Algergawy, M. Cheatham, D. Faria, A. Ferrara, I. Fundulaki, I. Harrow, S. Hertling,
E. Jime´nez-Ruiz, N. Karam, A. Khiat, P. Lambrix, H. Li, S. Montanelli, H. Paulheim,
C. Pesquita, T. Saveta, D. Schmidt, P. Shvaiko, A. Splendiani, E´. Thie´blin, C. Trojahn,
J. Vatascinova´, O. Zamazal, L. Zhou, Results of the ontology alignment evaluation initiative
2018, in: Proceedings of the 13th International Workshop on Ontology Matching, Monterey
(CA, US), 2018, pp. 76–116.
[10] M. Achichi, M. Cheatham, Z. Dragisic, J. Euzenat, D. Faria, A. Ferrara, G. Flouris, I.
Fundulaki, I. Harrow, V. Ivanova, E. Jime´nez-Ruiz, K. Kolthoff, E. Kuss, P. Lambrix, H. Leopold,
H. Li, C. Meilicke, M. Mohammadi, S. Montanelli, C. Pesquita, T. Saveta, P. Shvaiko,
A. Splendiani, H. Stuckenschmidt, E´. Thie´blin, K. Todorov, C. Trojahn, O. Zamazal,
Results of the ontology alignment evaluation initiative 2017, in: Proceedings of the 12th
International Workshop on Ontology Matching, Vienna, Austria, 2017, pp. 61–113. URL:
http://ceur-ws.org/Vol-2032/oaei17 paper0.pdf.
[11] M. Achichi, M. Cheatham, Z. Dragisic, J. Euzenat, D. Faria, A. Ferrara, G. Flouris, I.
Fundulaki, I. Harrow, V. Ivanova, E. Jime´nez-Ruiz, E. Kuss, P. Lambrix, H. Leopold, H. Li,
C. Meilicke, S. Montanelli, C. Pesquita, T. Saveta, P. Shvaiko, A. Splendiani, H.
Stuckenschmidt, K. Todorov, C. Trojahn, O. Zamazal, Results of the ontology alignment evaluation
initiative 2016, in: Proceedings of the 11th International Ontology matching workshop,
Kobe (JP), 2016, pp. 73–129.
[12] M. Cheatham, Z. Dragisic, J. Euzenat, D. Faria, A. Ferrara, G. Flouris, I. Fundulaki,
R. Granada, V. Ivanova, E. Jime´nez-Ruiz, P. Lambrix, S. Montanelli, C. Pesquita, T. Saveta,
P. Shvaiko, A. Solimando, C. Trojahn, O. Zamazal, Results of the ontology alignment
evaluation initiative 2015, in: Proceedings of the 10th International Ontology matching
workshop, Bethlehem (PA, US), 2015, pp. 60–115.
[13] Z. Dragisic, K. Eckert, J. Euzenat, D. Faria, A. Ferrara, R. Granada, V. Ivanova,
E. Jime´nez-Ruiz, A. O. Kempf, P. Lambrix, S. Montanelli, H. Paulheim, D. Ritze,
P. Shvaiko, A. Solimando, C. T. dos Santos, O. Zamazal, B. C. Grau, Results of
the ontology alignment evaluation initiative 2014, in: Proceedings of the 9th
International Ontology matching workshop, Riva del Garda (IT), 2014, pp. 61–104. URL:
http://ceur-ws.org/Vol-1317/oaei14 paper0.pdf.
[14] B. Cuenca Grau, Z. Dragisic, K. Eckert, J. Euzenat, A. Ferrara, R. Granada, V. Ivanova,
E. Jime´nez-Ruiz, A. Kempf, P. Lambrix, A. Nikolov, H. Paulheim, D. Ritze, F. Scharffe,
P. Shvaiko, C. Trojahn dos Santos, O. Zamazal, Results of the ontology alignment evaluation
initiative 2013, in: P. Shvaiko, J. Euzenat, K. Srinivas, M. Mao, E. Jime´nez-Ruiz (Eds.),
Proceedings of the 8th International Ontology matching workshop, Sydney (NSW, AU),
2013, pp. 61–100. URL: http://oaei.ontologymatching.org/2013/results/oaei2013.pdf.
[15] J. Aguirre, B. Cuenca Grau, K. Eckert, J. Euzenat, A. Ferrara, R. van Hague, L. Hollink,
E. Jime´nez-Ruiz, C. Meilicke, A. Nikolov, D. Ritze, F. Scharffe, P. Shvaiko, O.
Sva´bZamazal, C. Trojahn, B. Zapilko, Results of the ontology alignment evaluation initiative
2012, in: Proceedings of the 7th International Ontology matching workshop, Boston (MA,
US), 2012, pp. 73–115. URL: http://oaei.ontologymatching.org/2012/results/oaei2012.pdf.
[16] J. Euzenat, A. Ferrara, R. van Hague, L. Hollink, C. Meilicke, A. Nikolov, F. Scharffe,
P. Shvaiko, H. Stuckenschmidt, O. Sva´b-Zamazal, C. Trojahn dos Santos, Results of the
ontology alignment evaluation initiative 2011, in: Proceedings of the 6th International
Ontology matching workshop, Bonn (DE), 2011, pp. 85–110.
[17] J. Euzenat, A. Ferrara, C. Meilicke, A. Nikolov, J. Pane, F. Scharffe, P. Shvaiko, H.
Stuckenschmidt, O. Sva´b-Zamazal, V. Sva´tek, C. Trojahn dos Santos, Results of the ontology
alignment evaluation initiative 2010, in: Proceedings of the 5th International Ontology
matching workshop, Shanghai (CN), 2010, pp. 85–117. URL: http://oaei.ontologymatching.
org/2010/results/oaei2010.pdf.
[18] J. Euzenat, A. Ferrara, L. Hollink, A. Isaac, C. Joslyn, V. Malaise´, C. Meilicke, A. Nikolov,
J. Pane, M. Sabou, F. Scharffe, P. Shvaiko, V. Spiliopoulos, H. Stuckenschmidt, O.
Sva´bZamazal, V. Sva´tek, C. Trojahn dos Santos, G. Vouros, S. Wang, Results of the ontology
alignment evaluation initiative 2009, in: Proceedings of the 4th International Ontology
matching workshop, Chantilly (VA, US), 2009, pp. 73–126.
[19] C. Caracciolo, J. Euzenat, L. Hollink, R. Ichise, A. Isaac, V. Malaise´, C. Meilicke, J. Pane,
P. Shvaiko, H. Stuckenschmidt, O. Sva´b-Zamazal, V. Sva´tek, Results of the ontology
alignment evaluation initiative 2008, in: Proceedings of the 3rd Ontology matching workshop,
Karlsruhe (DE), 2008, pp. 73–120.
[20] J. Euzenat, A. Isaac, C. Meilicke, P. Shvaiko, H. Stuckenschmidt, O. Svab, V. Svatek,
W. van Hage, M. Yatskevich, Results of the ontology alignment evaluation initiative 2007,
in: Proceedings 2nd International Ontology matching workshop, Busan (KR), 2007, pp.
96–132. URL: http://ceur-ws.org/Vol-304/paper9.pdf.
[21] J. Euzenat, M. Mochol, P. Shvaiko, H. Stuckenschmidt, O. Svab, V. Svatek, W. R. van
Hage, M. Yatskevich, Results of the ontology alignment evaluation initiative 2006, in:
Proceedings of the 1st International Ontology matching workshop, Athens (GA, US), 2006,
pp. 73–95. URL: http://ceur-ws.org/Vol-225/paper7.pdf.
[22] S. Hertling, J. Portisch, H. Paulheim, Melt - matching evaluation toolkit, in: M. Acosta,
P. Cudre´-Mauroux, M. Maleshkova, T. Pellegrini, H. Sack, Y. Sure-Vetter (Eds.), Semantic
Systems. The Power of AI and Knowledge Graphs, Springer International Publishing, Cham,
2019, pp. 231–245.
friendly biomedical datasets for equivalence and subsumption ontology matching, in:
U. Sattler, A. Hogan, C. M. Keet, V. Presutti, J. P. A. Almeida, H. Takeda, P. Monnin,
G. Pirro`, C. d’Amato (Eds.), The Semantic Web - ISWC 2022 - 21st International Semantic
Web Conference, Virtual Event, October 23-27, 2022, Proceedings, volume 13489 of
Lecture Notes in Computer Science, Springer, 2022, pp. 575–591. URL: https://doi.org/10.
1007/978-3-031-19433-7 33. doi:10.1007/978-3-031-19433-7\_33.
[34] N. A. Vasilevsky, N. A. Matentzoglu, S. Toro, J. E. Flack IV, H. Hegde, D. R. Unni, G. F.</p>
      <p>Alyea, J. S. Amberger, L. Babb, J. P. Balhoff, et al., Mondo: Unifying diseases for the
world, by the world, medRxiv (2022) 2022–04.
[35] O. Bodenreider, The unified medical language system (umls): integrating biomedical
terminology, Nucleic acids research (2004).
[36] E. Jime´nez-Ruiz, B. C. Grau, LogMap: Logic-based and scalable ontology matching, in:
Proceedings of the 10th International Semantic Web Conference, Bonn (DE), 2011, pp.
273–288.
[37] Y. He, J. Chen, H. Dong, I. Horrocks, Exploring large language models for ontology
alignment, arXiv preprint arXiv:2309.07172 (2023).
[38] Y. He, J. Chen, H. Dong, I. Horrocks, C. Allocca, T. Kim, B. Sapkota, Deeponto: A python
package for ontology engineering with deep learning, arXiv preprint arXiv:2307.03067
(2023).
[39] N. Karam, C. Mu¨ller-Birn, M. Gleisberg, D. Fichtmu¨ller, R. Tolksdorf, A. Gu¨ntsch, A
terminology service supporting semantic annotation, integration, discovery and analysis
of interdisciplinary research data, Datenbank-Spektrum 16 (2016) 195–205. URL: https:
//doi.org/10.1007/s13222-016-0231-8. doi:10.1007/s13222-016-0231-8.
[40] F. Klan, E. Faessler, A. Algergawy, B. Ko¨nig-Ries, U. Hahn, Integrated semantic search
on structured and unstructured data in the adonis system, in: Proceedings of the 2nd
International Workshop on Semantics for Biodiversity, 2017.
[41] N. Karam, A. Khiat, A. Algergawy, M. Sattler, C. Weiland, M. Schmidt, Matching
biodiversity and ecology ontologies: challenges and evaluation results, Knowl. Eng.
Rev. 35 (2020) e9. URL: https://doi.org/10.1017/S0269888920000132. doi:10.1017/
S0269888920000132.
[42] F. Michel, O. Gargominy, S. Tercerie, C. Faron-Zucker, A Model to Represent
Nomenclatural and Taxonomic Information as Linked Data. Application to the French Taxonomic
Register, TAXREF, in: A. Algergawy, N. Karam, F. Klan, C. Jonquet (Eds.), Proceedings
of the 2nd International Workshop on Semantics for Biodiversity co-located with 16th
International Semantic Web Conference (ISWC 2017), Vienna, Austria, October 22nd,
2017, volume 1933 of CEUR Workshop Proceedings, CEUR-WS.org, 2017.
[43] A. Algergawy, N. Karam, A. Laadhar, F. Michel, Too big to match: a strategy around
matching tasks for large taxonomies, in: P. Shvaiko, J. Euzenat, E. Jime´nez-Ruiz, O.
Hassanzadeh, C. Trojahn (Eds.), Proceedings of the 17th International Workshop on
Ontology Matching (OM 2022) co-located with the 21th International Semantic Web
Conference (ISWC 2022), Hangzhou, China, held as a virtual conference, October 23, 2022,
volume 3324 of CEUR Workshop Proceedings, CEUR-WS.org, 2022, pp. 67–72. URL:
https://ceur-ws.org/Vol-3324/om2022 STpaper1.pdf.
[44] T. Ashino, Materials Ontology: An Infrastructure for Exchanging Materials Information</p>
    </sec>
  </body>
  <back>
    <ref-list>
      <ref id="ref1">
        <mixed-citation>
          [1]
          <string-name>
            <given-names>J.</given-names>
            <surname>Euzenat</surname>
          </string-name>
          ,
          <string-name>
            <given-names>C.</given-names>
            <surname>Meilicke</surname>
          </string-name>
          ,
          <string-name>
            <given-names>P.</given-names>
            <surname>Shvaiko</surname>
          </string-name>
          ,
          <string-name>
            <given-names>H.</given-names>
            <surname>Stuckenschmidt</surname>
          </string-name>
          ,
          <string-name>
            <surname>C.</surname>
          </string-name>
          <article-title>Trojahn dos Santos, Ontology alignment evaluation initiative: six years of experience</article-title>
          ,
          <source>Journal on Data Semantics XV</source>
          (
          <year>2011</year>
          )
          <fpage>158</fpage>
          -
          <lpage>192</lpage>
          .
        </mixed-citation>
      </ref>
      <ref id="ref2">
        <mixed-citation>
          [2]
          <string-name>
            <given-names>J.</given-names>
            <surname>Euzenat</surname>
          </string-name>
          ,
          <string-name>
            <given-names>P.</given-names>
            <surname>Shvaiko</surname>
          </string-name>
          , Ontology matching, 2nd ed., Springer-Verlag,
          <year>2013</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref3">
        <mixed-citation>
          [3]
          <string-name>
            <given-names>Y.</given-names>
            <surname>Sure</surname>
          </string-name>
          ,
          <string-name>
            <given-names>O.</given-names>
            <surname>Corcho</surname>
          </string-name>
          ,
          <string-name>
            <given-names>J.</given-names>
            <surname>Euzenat</surname>
          </string-name>
          , T. Hughes (Eds.),
          <source>Proceedings of the Workshop on Evaluation of Ontology-based Tools (EON)</source>
          ,
          <source>Hiroshima (JP)</source>
          ,
          <year>2004</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref4">
        <mixed-citation>
          [4]
          <string-name>
            <given-names>B.</given-names>
            <surname>Ashpole</surname>
          </string-name>
          ,
          <string-name>
            <given-names>M.</given-names>
            <surname>Ehrig</surname>
          </string-name>
          ,
          <string-name>
            <given-names>J.</given-names>
            <surname>Euzenat</surname>
          </string-name>
          , H. Stuckenschmidt (Eds.),
          <string-name>
            <surname>Proc.</surname>
          </string-name>
          K-Cap Workshop on Integrating Ontologies,
          <source>Banff (Canada)</source>
          ,
          <year>2005</year>
          . URL: http://ceur-ws.
          <source>org/</source>
          Vol-
          <volume>156</volume>
          /.
        </mixed-citation>
      </ref>
      <ref id="ref5">
        <mixed-citation>
          [5]
          <string-name>
            <given-names>M.</given-names>
            <surname>Abd Nikooie Pour</surname>
          </string-name>
          ,
          <string-name>
            <given-names>A.</given-names>
            <surname>Algergawy</surname>
          </string-name>
          ,
          <string-name>
            <given-names>P.</given-names>
            <surname>Buche</surname>
          </string-name>
          ,
          <string-name>
            <given-names>L. J.</given-names>
            <surname>Castro</surname>
          </string-name>
          ,
          <string-name>
            <given-names>J.</given-names>
            <surname>Chen</surname>
          </string-name>
          ,
          <string-name>
            <given-names>H.</given-names>
            <surname>Dong</surname>
          </string-name>
          ,
          <string-name>
            <given-names>O.</given-names>
            <surname>Fallatah</surname>
          </string-name>
          ,
          <string-name>
            <given-names>D.</given-names>
            <surname>Faria</surname>
          </string-name>
          , I. Fundulaki,
          <string-name>
            <given-names>S.</given-names>
            <surname>Hertling</surname>
          </string-name>
          ,
          <string-name>
            <given-names>Y.</given-names>
            <surname>He</surname>
          </string-name>
          ,
          <string-name>
            <surname>I. Horrocks</surname>
          </string-name>
          ,
          <string-name>
            <given-names>M.</given-names>
            <surname>Huschka</surname>
          </string-name>
          ,
          <string-name>
            <given-names>L.</given-names>
            <surname>Ibanescu</surname>
          </string-name>
          ,
          <string-name>
            <surname>E.</surname>
          </string-name>
          <article-title>Jime´nez-</article-title>
          <string-name>
            <surname>Ruiz</surname>
            ,
            <given-names>N.</given-names>
          </string-name>
          <string-name>
            <surname>Karam</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          <string-name>
            <surname>Laadhar</surname>
            ,
            <given-names>P.</given-names>
          </string-name>
          <string-name>
            <surname>Lambrix</surname>
            ,
            <given-names>H.</given-names>
          </string-name>
          <string-name>
            <surname>Li</surname>
            ,
            <given-names>Y.</given-names>
          </string-name>
          <string-name>
            <surname>Li</surname>
            ,
            <given-names>F.</given-names>
          </string-name>
          <string-name>
            <surname>Michel</surname>
            , E. Nasr,
            <given-names>H.</given-names>
          </string-name>
          <string-name>
            <surname>Paulheim</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          <string-name>
            <surname>Pesquita</surname>
            ,
            <given-names>T.</given-names>
          </string-name>
          <string-name>
            <surname>Saveta</surname>
            ,
            <given-names>P.</given-names>
          </string-name>
          <string-name>
            <surname>Shvaiko</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          <string-name>
            <surname>Trojahn</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          <string-name>
            <surname>Verhey</surname>
            ,
            <given-names>M.</given-names>
          </string-name>
          <string-name>
            <surname>Wu</surname>
            ,
            <given-names>B.</given-names>
          </string-name>
          <string-name>
            <surname>Yaman</surname>
            ,
            <given-names>O.</given-names>
          </string-name>
          <string-name>
            <surname>Zamazal</surname>
            ,
            <given-names>L.</given-names>
          </string-name>
          <string-name>
            <surname>Zhou</surname>
          </string-name>
          ,
          <article-title>Results of the ontology alignment evaluation initiative 2022</article-title>
          , in: P. Shvaiko,
          <string-name>
            <given-names>J.</given-names>
            <surname>Euzenat</surname>
          </string-name>
          ,
          <string-name>
            <surname>E.</surname>
          </string-name>
          <article-title>Jime´nez-</article-title>
          <string-name>
            <surname>Ruiz</surname>
            ,
            <given-names>O.</given-names>
          </string-name>
          <string-name>
            <surname>Hassanzadeh</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          Trojahn (Eds.),
          <source>Proceedings of the 17th International Workshop on Ontology Matching (OM</source>
          <year>2022</year>
          )
          <article-title>co-located with the 21th International Semantic Web Conference (ISWC 2022), Hangzhou, China, held as a virtual conference</article-title>
          ,
          <source>October</source>
          <volume>23</volume>
          ,
          <year>2022</year>
          , volume
          <volume>3324</volume>
          <source>of CEUR Workshop Proceedings, CEUR-WS.org</source>
          ,
          <year>2022</year>
          , pp.
          <fpage>84</fpage>
          -
          <lpage>128</lpage>
          . URL: https://ceur-ws.
          <source>org/</source>
          Vol-
          <volume>3324</volume>
          /oaei22 paper0.pdf.
        </mixed-citation>
      </ref>
      <ref id="ref6">
        <mixed-citation>
          [6]
          <string-name>
            <given-names>M.</given-names>
            <surname>Abd Nikooie Pour</surname>
          </string-name>
          ,
          <string-name>
            <given-names>A.</given-names>
            <surname>Algergawy</surname>
          </string-name>
          ,
          <string-name>
            <given-names>F.</given-names>
            <surname>Amardeilh</surname>
          </string-name>
          ,
          <string-name>
            <given-names>R.</given-names>
            <surname>Amini</surname>
          </string-name>
          ,
          <string-name>
            <given-names>O.</given-names>
            <surname>Fallatah</surname>
          </string-name>
          ,
          <string-name>
            <given-names>D.</given-names>
            <surname>Faria</surname>
          </string-name>
          ,
          <string-name>
            <surname>I. Fundulaki</surname>
          </string-name>
          , I. Harrow,
          <string-name>
            <given-names>S.</given-names>
            <surname>Hertling</surname>
          </string-name>
          ,
          <string-name>
            <given-names>P.</given-names>
            <surname>Hitzler</surname>
          </string-name>
          ,
          <string-name>
            <given-names>M.</given-names>
            <surname>Huschka</surname>
          </string-name>
          ,
          <string-name>
            <given-names>L.</given-names>
            <surname>Ibanescu</surname>
          </string-name>
          ,
          <string-name>
            <surname>E.</surname>
          </string-name>
          <article-title>Jime´nez-</article-title>
          <string-name>
            <surname>Ruiz</surname>
            ,
            <given-names>N.</given-names>
          </string-name>
          <string-name>
            <surname>Karam</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          <string-name>
            <surname>Laadhar</surname>
            ,
            <given-names>P.</given-names>
          </string-name>
          <string-name>
            <surname>Lambrix</surname>
            ,
            <given-names>H.</given-names>
          </string-name>
          <string-name>
            <surname>Li</surname>
            ,
            <given-names>Y.</given-names>
          </string-name>
          <string-name>
            <surname>Li</surname>
            ,
            <given-names>F.</given-names>
          </string-name>
          <string-name>
            <surname>Michel</surname>
            , E. Nasr,
            <given-names>H.</given-names>
          </string-name>
          <string-name>
            <surname>Paulheim</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          <string-name>
            <surname>Pesquita</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          <string-name>
            <surname>Portisch</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          <string-name>
            <surname>Roussey</surname>
            ,
            <given-names>T.</given-names>
          </string-name>
          <string-name>
            <surname>Saveta</surname>
            ,
            <given-names>P.</given-names>
          </string-name>
          <string-name>
            <surname>Shvaiko</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          <string-name>
            <surname>Splendiani</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          <string-name>
            <surname>Trojahn</surname>
            , J. Vatascinova´,
            <given-names>B.</given-names>
          </string-name>
          <string-name>
            <surname>Yaman</surname>
            ,
            <given-names>O.</given-names>
          </string-name>
          <string-name>
            <surname>Zamazal</surname>
            ,
            <given-names>L.</given-names>
          </string-name>
          <string-name>
            <surname>Zhou</surname>
          </string-name>
          ,
          <article-title>Results of the ontology alignment evaluation initiative 2021</article-title>
          , in: P. Shvaiko,
          <string-name>
            <given-names>J.</given-names>
            <surname>Euzenat</surname>
          </string-name>
          ,
          <string-name>
            <surname>E.</surname>
          </string-name>
          <article-title>Jime´nez-</article-title>
          <string-name>
            <surname>Ruiz</surname>
            ,
            <given-names>O.</given-names>
          </string-name>
          <string-name>
            <surname>Hassanzadeh</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          Trojahn (Eds.),
          <source>Proceedings of the 16th International Workshop on Ontology Matching co-located with the 20th International Semantic Web Conference (ISWC</source>
          <year>2021</year>
          ), Virtual conference,
          <source>October</source>
          <volume>25</volume>
          ,
        </mixed-citation>
      </ref>
    </ref-list>
  </back>
</article>