<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD v1.0 20120330//EN" "JATS-archivearticle1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink">
  <front>
    <journal-meta>
      <journal-title-group>
        <journal-title>Journal of Biomedical Semantics 8 (2017) 56:1-56:28. doi:10.
1186/s13326</journal-title>
      </journal-title-group>
    </journal-meta>
    <article-meta>
      <article-id pub-id-type="doi">10.1093/database/baac035</article-id>
      <title-group>
        <article-title>Results of the Ontology Alignment Evaluation Initiative 2024</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="author">
          <string-name>Mina Abd Nikooie Pour</string-name>
          <xref ref-type="aff" rid="aff18">18</xref>
          <xref ref-type="aff" rid="aff8">8</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Alsayed Algergawy</string-name>
          <xref ref-type="aff" rid="aff10">10</xref>
          <xref ref-type="aff" rid="aff2">2</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Eva Blomqvist</string-name>
          <xref ref-type="aff" rid="aff8">8</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Patrice Buche</string-name>
          <xref ref-type="aff" rid="aff20">20</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Jiaoyan Chen</string-name>
          <xref ref-type="aff" rid="aff5">5</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Pedro Giesteira Cotovio</string-name>
          <xref ref-type="aff" rid="aff15">15</xref>
          <xref ref-type="aff" rid="aff3">3</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Adrien Coulet</string-name>
          <xref ref-type="aff" rid="aff1">1</xref>
          <xref ref-type="aff" rid="aff12">12</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Julien Cu</string-name>
          <xref ref-type="aff" rid="aff20">20</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Hang Dong</string-name>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Daniel Faria</string-name>
          <xref ref-type="aff" rid="aff11">11</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Lucas Ferraz</string-name>
          <xref ref-type="aff" rid="aff15">15</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Sven Hertling</string-name>
          <xref ref-type="aff" rid="aff4">4</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Yuan He</string-name>
          <xref ref-type="aff" rid="aff7">7</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Ian Horrocks</string-name>
          <xref ref-type="aff" rid="aff7">7</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Liliana Ibanescu</string-name>
          <xref ref-type="aff" rid="aff23">23</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Sarika Jain</string-name>
          <xref ref-type="aff" rid="aff16">16</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Ernesto Jime´nez-Ruiz</string-name>
          <xref ref-type="aff" rid="aff3">3</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Naouel Karam</string-name>
          <xref ref-type="aff" rid="aff13">13</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Felix Kraus</string-name>
          <xref ref-type="aff" rid="aff14">14</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Patrick Lambrix</string-name>
          <xref ref-type="aff" rid="aff18">18</xref>
          <xref ref-type="aff" rid="aff8">8</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Huanyu Li</string-name>
          <xref ref-type="aff" rid="aff8">8</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Ying Li</string-name>
          <xref ref-type="aff" rid="aff18">18</xref>
          <xref ref-type="aff" rid="aff8">8</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Pierre Monnin</string-name>
          <xref ref-type="aff" rid="aff22">22</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Heiko Paulheim</string-name>
          <xref ref-type="aff" rid="aff4">4</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Catia Pesquita</string-name>
          <xref ref-type="aff" rid="aff15">15</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Abhisek Sharma</string-name>
          <xref ref-type="aff" rid="aff16">16</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Pavel Shvaiko</string-name>
          <xref ref-type="aff" rid="aff19">19</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Marta Silva</string-name>
          <xref ref-type="aff" rid="aff15">15</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Guilherme Sousa</string-name>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Cassia Trojahn</string-name>
          <xref ref-type="aff" rid="aff21">21</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Jana Vatasˇcˇinova´</string-name>
          <xref ref-type="aff" rid="aff17">17</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Beyza Yaman</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Ondrˇej Zamazal</string-name>
          <xref ref-type="aff" rid="aff17">17</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Lu Zhou</string-name>
          <xref ref-type="aff" rid="aff9">9</xref>
        </contrib>
        <aff id="aff0">
          <label>0</label>
          <institution>ADAPT Centre, Trinity College Dublin</institution>
        </aff>
        <aff id="aff1">
          <label>1</label>
          <institution>Centre de Recherche des Cordeliers, Inserm, Universite ́ Paris Cite ́, Sorbonne Universite ́</institution>
          ,
          <country country="FR">France</country>
        </aff>
        <aff id="aff2">
          <label>2</label>
          <institution>Chair of Data and Knowledge Engineering, University of Passau</institution>
          ,
          <country country="DE">Germany</country>
        </aff>
        <aff id="aff3">
          <label>3</label>
          <institution>City St George's, University of London, UK &amp; University of Oslo</institution>
          ,
          <country country="NO">Norway</country>
        </aff>
        <aff id="aff4">
          <label>4</label>
          <institution>Data and Web Science Group, University of Mannheim</institution>
          ,
          <country country="DE">Germany</country>
        </aff>
        <aff id="aff5">
          <label>5</label>
          <institution>Department of Computer Science, The University of Manchester</institution>
          ,
          <country country="UK">UK</country>
        </aff>
        <aff id="aff6">
          <label>6</label>
          <institution>Department of Computer Science, University of Exeter</institution>
          ,
          <country country="UK">UK</country>
        </aff>
        <aff id="aff7">
          <label>7</label>
          <institution>Department of Computer Science, University of Oxford</institution>
          ,
          <country country="UK">UK</country>
        </aff>
        <aff id="aff8">
          <label>8</label>
          <institution>Department of Computer and Information Science, Linko ̈ping University</institution>
          ,
          <addr-line>Linko ̈ping</addr-line>
          ,
          <country country="SE">Sweden</country>
        </aff>
        <aff id="aff9">
          <label>9</label>
          <institution>Flatfee Corp</institution>
          ,
          <country country="US">USA</country>
        </aff>
        <aff id="aff10">
          <label>10</label>
          <institution>Heinz Nixdorf Chair for Distributed Information Systems, Friedrich Schiller University Jena</institution>
          ,
          <country country="DE">Germany</country>
        </aff>
        <aff id="aff11">
          <label>11</label>
          <institution>INESC-ID / IST, University of Lisbon</institution>
          ,
          <country country="PT">Portugal</country>
        </aff>
        <aff id="aff12">
          <label>12</label>
          <institution>Inria Paris</institution>
          ,
          <country country="FR">France</country>
        </aff>
        <aff id="aff13">
          <label>13</label>
          <institution>Institute for Applied Informatics, University of Leipzig</institution>
          ,
          <country country="DE">Germany</country>
        </aff>
        <aff id="aff14">
          <label>14</label>
          <institution>Karlsruhe Institute of Technology</institution>
          ,
          <addr-line>Karlsruhe</addr-line>
          ,
          <country country="DE">Germany</country>
        </aff>
        <aff id="aff15">
          <label>15</label>
          <institution>LASIGE</institution>
          ,
          <addr-line>Faculdade de Cie</addr-line>
        </aff>
        <aff id="aff16">
          <label>16</label>
          <institution>National Institute of Technology Kurukshetra</institution>
          ,
          <addr-line>Haryana</addr-line>
          ,
          <country country="IN">India</country>
        </aff>
        <aff id="aff17">
          <label>17</label>
          <institution>Prague University of Economics and Business</institution>
          ,
          <country country="CZ">Czech Republic</country>
        </aff>
        <aff id="aff18">
          <label>18</label>
          <institution>Swedish e-Science Research Centre</institution>
          ,
          <addr-line>Linko ̈ping</addr-line>
          ,
          <country country="SE">Sweden</country>
        </aff>
        <aff id="aff19">
          <label>19</label>
          <institution>Trentino Digitale SpA</institution>
          ,
          <addr-line>Trento</addr-line>
          ,
          <country country="IT">Italy</country>
        </aff>
        <aff id="aff20">
          <label>20</label>
          <institution>UMR IATE, INRAE, University of Montpellier</institution>
          ,
          <country country="FR">France</country>
        </aff>
        <aff id="aff21">
          <label>21</label>
          <institution>Univ. Grenoble Alpes</institution>
          ,
          <addr-line>Inria, CNRS, Grenoble INP, LIG, F-38000 Grenoble</addr-line>
          ,
          <country country="FR">France</country>
        </aff>
        <aff id="aff22">
          <label>22</label>
          <institution>Universite ́ Co</institution>
        </aff>
        <aff id="aff23">
          <label>23</label>
          <institution>Universite ́ Paris-Saclay</institution>
          ,
          <addr-line>INRAE, AgroParisTech, UMR MIA Paris-Saclay</addr-line>
          ,
          <country country="FR">France</country>
        </aff>
        <aff id="aff24">
          <label>24</label>
          <institution>te d'Azur</institution>
          ,
          <addr-line>Inria, CNRS, I3S, Sophia Antipolis</addr-line>
          ,
          <country country="FR">France</country>
        </aff>
      </contrib-group>
      <pub-date>
        <year>2024</year>
      </pub-date>
      <volume>2788</volume>
      <fpage>92</fpage>
      <lpage>138</lpage>
      <abstract>
        <p>The Ontology Alignment Evaluation Initiative (OAEI) aims at comparing ontology matching systems on precisely dened test cases. These test cases can be based on ontologies of dierent levels of complexity and use dierent evaluation modalities. The OAEI 2024 campaign oered 13 tracks and was attended by 13 participants. This paper is an overall presentation of that campaign.</p>
      </abstract>
    </article-meta>
  </front>
  <body>
    <sec id="sec-1">
      <title>1. Introduction</title>
      <p>
        The Ontology Alignment Evaluation Initiative1 (OAEI) is a coordinated international initiative, which
organizes the evaluation of ontology matching systems [
        <xref ref-type="bibr" rid="ref1 ref2">1, 2</xref>
        ], and has been run for 20 years now. The
main goal of the OAEI is to compare systems and algorithms openly and on the same basis to allow
anyone to conclude the best ontology matching strategies. Furthermore, the ambition is that from
such evaluations, developers can improve their systems and oer better tools addressing the evolving
application needs.
      </p>
      <p>
        The rst two events were organized in 2004: (i) the Information Interpretation and Integration
Conference (I3CON) held at the NIST Performance Metrics for Intelligent Systems (PerMIS) workshop
and (ii) the Ontology Alignment Contest held at the Evaluation of Ontology-based Tools (EON) workshop
of the annual International Semantic Web Conference (ISWC) [
        <xref ref-type="bibr" rid="ref3">3</xref>
        ]. Then, a unique OAEI campaign
occurred in 2005 at the workshop on Integrating Ontologies held in conjunction with the International
Conference on Knowledge Capture (K-Cap) [
        <xref ref-type="bibr" rid="ref4">4</xref>
        ]. From 2006 until the present, the OAEI campaigns were
held at the Ontology Matching workshop, co-located with ISWC [
        <xref ref-type="bibr" rid="ref5 ref6 ref7">5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16,
17, 18, 19, 20, 21, 22</xref>
        ], which this year took place in Baltimore, USA2.
      </p>
      <p>Since 2011, we have been using an environment for automatically processing evaluations that was
developed within the SEALS (Semantic Evaluation At Large Scale) project3. SEALS provided a soware
infrastructure to automatically execute evaluations and evaluation campaigns for typical semantic web
tools, including ontology matching. Since OAEI 2017, a novel evaluation environment called HOBBIT
(Section 2.1) was adopted for the HOBBIT Link Discovery track, and later extended to enable the
evaluation of other tracks. Some tracks are run exclusively through SEALS and others through HOBBIT,
but several allow participants to choose their preferred platform. Since 2022, the MELT framework
[23] has been adopted to facilitate the SEALS and HOBBIT wrapping and evaluation. Since 2023, most
tracks have adopted MELT as their evaluation platform.</p>
      <p>This paper synthesizes the 2024 evaluation campaign and introduces the results provided in the
participants’ papers. The remainder of the paper is organized as follows: in Section 2, we present the
overall evaluation methodology; in Section 3, we present the tracks and datasets; in Section 4 we present
and discuss the results; and nally, Section 5 discusses the lessons learned.</p>
    </sec>
    <sec id="sec-2">
      <title>2. Methodology</title>
      <sec id="sec-2-1">
        <title>2.1. Evaluation platforms</title>
        <p>The OAEI evaluation was conducted in one of three alternative platforms: the SEALS client, the
HOBBIT platform, or the MELT framework. All of them have the goal of ensuring reproducibility
and comparability of the results across matching systems. As of this campaign, the use of the SEALS
client and packaging format is deprecated in favor of MELT, with the sole exception of the Interactive
Matching track, as simulated interactive matching is not yet supported by MELT.</p>
        <p>The SEALS client was developed in 2011. It is a Java-based command line interface for ontology
matching evaluation, which requires system developers to implement an interface and to wrap their
tools in a predened way, including all required libraries and resources.</p>
        <p>The HOBBIT platform4 was introduced in 2017. It is a web interface for linked data and ontology
matching evaluation, which requires systems to be wrapped inside docker containers and includes a
SystemAdapter class, then being uploaded into the HOBBIT platform [24].</p>
        <p>The MELT framework5 [23] was introduced in 2019 and is under active development. It allows
the development, evaluation, and packaging of matching systems for evaluation interfaces like SEALS
1http://oaei.ontologymatching.org
2http://om.ontologymatching.org/2024
3http://www.seals-project.eu
4https://project-hobbit.eu/outcomes/hobbit-platform/
5https://github.com/dwslab/melt
or HOBBIT. It further enables developers to use Python or any other programming language in their
matching systems, which beforehand had been a hurdle for OAEI participants. The evaluation client6
allows organizers to evaluate packaged systems whereby multiple submission formats are supported
(SEALS packages or matchers implemented as Web services). Starting with this year, the MELT
framework also supports the SSSOM [25] format. Therefore, systems producing an alignment in the
SSSOM format can be evaluated as well.</p>
        <p>All platforms compute the standard evaluation metrics against the reference alignments: precision,
recall, and F-measure. In test cases requiring dierent evaluation modalities, the evaluation was carried
out a posteriori, using the alignments produced by the matching systems.</p>
      </sec>
      <sec id="sec-2-2">
        <title>2.2. Submission formats</title>
        <p>As already mentioned above, three submission formats were allowed: (1) SEALS package, (2) HOBBIT,
and (3) MELT. With the increasing usage of other programming languages than Java and increasing
hardware requirements for matching systems, since 2021 the MELT Web interface was introduced to
address this issue. It mainly consists of a technology-independent HTTP interface7 which participants
can implement as they wish. Alternatively, they can use the MELT framework to assist them, as it can
be used to wrap any matching system as a docker container that implements the HTTP interface.</p>
        <p>In this year, we also allowed to submit alignment les in addition to the executable system in case it
requires substantial hardware or soware resources.</p>
      </sec>
      <sec id="sec-2-3">
        <title>2.3. OAEI campaign phases</title>
        <p>As in previous years, the OAEI 2024 campaign was divided into three phases: preparatory, execution,
and evaluation.</p>
        <p>In the preparation phase, the test cases were provided to participants during an initial evaluation
period between June 30ℎ and July 31, 2024. The goal of this phase is to ensure that the test cases
make sense to the participants and give them the opportunity to provide feedback to organizers on the
test case, as well as potentially report errors. At the end of this phase, the nal test base was frozen and
released.</p>
        <p>During the subsequent execution phase, participants test and potentially develop their matching
systems to automatically match the test cases. Participants can self-evaluate their results either by
comparing their output with the reference alignments or by using either of the evaluation platforms.
They can tune their systems with respect to the non-blind evaluation as long as they respect the rules
of the OAEI. Participants were required to register their systems by July 31 and make a preliminary
evaluation by August 31. The execution phase was terminated on September 30ℎ, 2024, at which
date participants had to submit the (near) nal versions of their systems (SEALS-wrapped and/or
HOBBIT-wrapped).</p>
        <p>During the evaluation phase, systems were evaluated by all track organizers. In case minor problems
were found during the initial stages of this phase, they were reported to the developers, who were
given the opportunity to x and resubmit their systems. Initial results were provided directly to the
participants, whereas nal results for most tracks were published on the respective OAEI web pages
before the workshop.</p>
      </sec>
    </sec>
    <sec id="sec-3">
      <title>3. Tracks and Test Cases</title>
      <p>This year’s OAEI campaign consisted of 13 tracks, all of them including OWL ontologies while only
one also including SKOS thesauri. They can be grouped into:</p>
      <p>– Schema matching tracks, which have as objective matching ontology classes and/or properties.
6https://dwslab.github.io/melt/matcher-evaluation/client
7https://dwslab.github.io/melt/matcher-packaging/web</p>
      <p>The tracks are summarized in Table 1 and detailed in the following sections.</p>
      <sec id="sec-3-1">
        <title>3.1. Anatomy</title>
        <p>The anatomy track comprises a single test case consisting of matching two fragments of biomedical
ontologies which describe the human anatomy8 (3304 classes) and the anatomy of the mouse9 (2744
classes). The evaluation is based on a manually curated reference alignment. This dataset has been used
since 2007 with some improvements over the years [26].</p>
        <p>Systems are evaluated with the standard parameters of precision, recall, F-measure. Additionally,
recall+ is computed by excluding trivial correspondences (i.e., correspondences that have the same
normalized label). Alignments are also checked for coherence using the Pellet reasoner. The evaluation
was carried out on a machine with a 5 core CPU @ 1.80 GHz with 16GB allocated RAM, using the
MELT framework. For some systems, the SEALS client has been used. However, the evaluation
parameters were computed a posteriori, aer removing from the alignments produced by the systems,
correspondences expressing relations other than equivalence, as well as trivial correspondences in the
oboInOwl namespace (e.g., oboInOwl#Synonym = oboInOwl#Synonym). The results obtained with the
SEALS client vary in some cases by 0.5% compared to the results presented in Section 4.2.
8https://www.cancer.gov/cancertopics/cancerlibrary/terminologyresources
9http://www.informatics.jax.org/searches/AMA form.shtml
relations
confidence modalities language SEALS HOBBIT MELT</p>
      </sec>
      <sec id="sec-3-2">
        <title>3.2. Conference</title>
        <p>The conference track consists of a suite of 21 matching tasks corresponding to the pairwise combination
of 7 moderately expressive ontologies describing the domain of organizing conferences. The dataset
and its usage is described in [27].</p>
        <p>The track uses several reference alignments for evaluation: the old (and not fully complete) manually
curated open reference alignment, ra1; an extended, also manually curated version of this alignment,
ra2; a version of the latter corrected to resolve violations of conservativity, rar2; and an uncertain
version of ra1 produced through crowd-sourcing, where the score of each correspondence is the fraction
of people in the evaluation group that agree with the correspondence. The latter reference was used in
two evaluation modalities: discrete and continuous evaluation. In the former, correspondences in the
uncertain reference alignment with a score of at least 0.5 are treated as correct whereas those with
lower score are treated as incorrect, and standard evaluation parameters are used to evaluated systems.
In the latter, weighted precision, recall and F-measure values are computed by taking into consideration
the actual scores of the uncertain reference, as well as the scores generated by the matching system. For
the sharp reference alignments (ra1, ra2 and rar2), the evaluation is based on the standard parameters,
as well the F0.5-measure and F2-measure and on conservativity and consistency violations. Whereas F1
is the harmonic mean of precision and recall where both receive equal weight, 2 gives higher weight
to recall than precision and F0.5 gives higher weight to precision higher than recall. The second test
case contains open reference alignment and systems were evaluated using the standard metrics.</p>
        <p>Two baseline matchers are used to benchmark the systems: edna string edit distance matcher; and
StringEquiv string equivalence matcher as in the anatomy test case.</p>
      </sec>
      <sec id="sec-3-3">
        <title>3.3. Multifarm</title>
        <p>The multifarm track [28] aims at evaluating the ability of matching systems to deal with ontologies in
dierent natural languages. This dataset results from the translation of 7 ontologies from the conference
track (cmt, conference, confOf, iasted, sigkdd, ekaw and edas) into 10 languages: Arabic (ar), Chinese
(cn), Czech (cz), Dutch (nl), French (fr), German (de), Italian (it), Portuguese (pt), Russian (ru), and
Spanish (es). The dataset is composed of 55 pairs of languages, with 49 matching tasks for each of
them, taking into account the alignment direction (e.g. cmt →edas and cmt →edas are distinct
matching tasks). While part of the dataset is openly available, all matching tasks involving the edas and
ekaw ontologies (resulting in 55 × 24 matching tasks) are used for blind evaluation.</p>
        <p>We consider two test cases: i) those tasks where two dierent ontologies (cmt→edas, for instance) have
been translated into two dierent languages; and ii) those tasks where the same ontology (cmt→cmt)
has been translated into two dierent languages. For the tasks of type ii), good results are not only
related to the use of specic techniques for dealing with cross-lingual ontologies, but also on the ability
to exploit the identical structure of the ontologies. This year, we report the results on dierent ontologies
(i).</p>
        <p>The reference alignments used in this track derive directly from the manually curated Conference
ra1 reference alignments. In 2021, alignments have been manually evaluated by domain experts. The
evaluation is blind. The systems have been executed on a Ubuntu Linux machine congured with
32GB of RAM running under a Intel Core CPU 2.00GHz x8 cores. The evaluation was performed using
the MELT platform. Every participating system was executed in its standard setting and we compare
precision, recall and F-measure as well as the computation time.</p>
      </sec>
      <sec id="sec-3-4">
        <title>3.4. Complex Matching</title>
        <p>The complex matching track is meant to evaluate the matchers based on their ability to generate complex
alignments. A complex alignment is composed of complex correspondences typically involving more
than two ontology entities, such as 1:AcceptedPaper ≡ 2:Paper ⊓ 2:hasDecision.2:Acceptance.</p>
        <p>As last year, the track run with two data sets from the conference domain: Conference and
Populated Conference, as the other complex sub-tracks (Hydrography, GeoLink, Populated GeoLink
Populated Enslaved, and Taxon datasets) have been discontinued.</p>
        <p>The Conference dataset comprises three ontologies: cmt, conference, and ekaw from the conference
dataset. The reference alignment was created as a consensus between experts. To allow matchers which
rely on instances to participate over the Conference complex track, the Populated Conference data
set is composed of 5 conference ontologies populated with more or less common instances, resulting in
6 datasets: (6 versions on the repository: v0, v20, v40, v60, v80 and v100). Details on the population and
evaluation modalities are available10.</p>
        <p>The participants of the track output their (complex) correspondences in the EDOAL format. For the
Conference dataset, the complex correspondences are manually compared to the ones of the consensus
alignment. For the Populated Conference dataset, the alignments are evaluated using coverage and
precision metrics using an evaluator that relies on the comparison of sets of instances [29]. All our
evaluations were conducted on a server machine with AMD EPYC 7402 2.8 GHz x48 processors, 512GB
RAM. Processes needing a GPU were run in a compute node with 4 Nvidia Geforce GTX 1080TI.
3.5. Food
The Food Nutritional Composition track aims at nding alignments between food concepts from
CIQUAL11, the French food nutritional composition database, and food concepts from SIREN12, the
Scientic Information and Retrieval Exchange Network of the US Food and Drug administration. Foods
from both databases are described in LanguaL13, a well-known multilingual thesaurus using faceted
classication. LanguaL stands for “Langua aLimentaria” or “language of food”; more than 40,000 foods
used in food composition databases are described using LanguaL.</p>
        <p>In [30], a method to provide OWL modelling of food concepts from both datasets, CIQUAL14 and
SIREN 15, and a gold standard are presented.</p>
        <p>The evaluation was performed using the MELT platform. Every participating system was executed
in its standard setting and we compare precision, recall and F-measure as well as the computation time.</p>
      </sec>
      <sec id="sec-3-5">
        <title>3.6. Interactive Matching</title>
        <p>The interactive matching track aims to assess the performance of semi-automated matching systems by
simulating user interaction [31, 32, 33]. The evaluation thus focuses on how interaction with the user
improves the matching results. Currently, this track does not evaluate the user experience or the user
interfaces of the systems [34, 32].</p>
        <p>The interactive matching track is based on the datasets from the Anatomy and Conference tracks,
which have been previously described. It relies on the SEALS client’s Oracle class to simulate user
interactions. An interactive matching system can present a collection of correspondences simultaneously
to the oracle, telling the system whether that correspondence is correct or not. If a system presents up
to three correspondences together and each correspondence presented has a mapped entity (i.e., class
or property) in common with at least one other correspondence presented, the oracle counts this as
a single interaction, under the rationale that this corresponds to a scenario where a user is asked to
choose between conicting candidate correspondences. To simulate the possibility of user errors, the
oracle can be set to reply with a given error probability (randomly, from a uniform distribution). We
evaluated systems with four dierent error rates: 0.0 (perfect user), 0.1, 0.2, and 0.3.</p>
        <p>In addition to the standard evaluation parameters, we also compute the number of requests made by
the system, the total number of distinct correspondences asked, the number of positive and negative
answers from the oracle, the performance of the system according to the oracle (to assess the impact of
10https://framagit.org/IRIT UT2J/conference-dataset-population
11https://ciqual.anses.fr/
12http://langual.org/langual indexed datasets.asp
13https://www.langual.org/default.asp
14https://entrepot.recherche.data.gouv.fr/dataset.xhtml?persistentId=doi:10.15454/6CEYU3
15https://entrepot.recherche.data.gouv.fr/dataset.xhtml?persistentId=doi:10.15454/5LLGVY
the oracle errors on the system) and nally, the performance of the oracle itself (to assess how erroneous
it was).</p>
        <p>The evaluation was carried out on a server with 3.46 GHz (6 cores) and 8GB RAM allocated to the
matching systems. For systems requiring more RAM, the evaluation was carried out on a computer
with an AMD Ryzen 7 5700G 3.80 GHz CPU and 32GB RAM, with 10GB of max heap space allocated
to java.Each system was run ten times and the nal result of a system for each error rate represents
the average of these runs. For the Conference dataset with the ra1 alignment, precision and recall
correspond to the micro-average over all ontology pairs, whereas the number of interactions is the total
number of interactions for all the pairs.
3.7. Bio-ML
The Bio-ML track [35] incorporates both equivalence and subsumption ontology matching (OM) tasks
for biomedical ontologies, with ground truth (equivalence) mappings extracted from Mondo [36] and
UMLS [37] (see Table 2). Mondo aims to integrate disease concepts worldwide, while UMLS is a
metathesaurus for the biomedical domain. Based on techniques (ontology pruning, subsumption mapping
construction, negative candidate mapping generation, etc.) proposed in [35], we make available ve OM
pairs with their information reported in Table 3. Each OM pair is accompanied with both equivalence
and subsumption matching tasks; each matching task has two data split settings, i.e., unsupervised
setting with no training mappings, and semi-supervised setting with 30% ground truth mappings for
training/validation. In addition, the Bio-LLM sub-track supports a more ecient evaluation of large
language model-based OM [38], which consists of challenging subsets of NCIT-DOID and
SNOMEDFMA (Body) datasets, along with tailored evaluation metrics. Since the 2023 edition, Bio-ML has added
a logical module enrichment [39] to add entities to the pruned ontologies to provide more context
for alignment, annotated as “not used in alignment” and ignored in evaluation. For evaluation, in
[35] we proposed both global matching and local ranking; the former aims to evaluate the overall
performance by computing Precision, Recall, and F1 metrics for the output mappings against the
reference mappings, while the latter aims to evaluate the ability to distinguish the correct mapping out
of several challenging negatives by ranking metrics Hits@K and MRR. Note that subsumption mappings
are inherently incomplete, so only local ranking evaluation is applied for subsumption matching. For
the special sub-track Bio-LLM, both matching and ranking metrics are used, but they are tailored to
the subsets, along with an additional metric called rejection rate to examine if systems can reject all
plausible mappings for entities that actually have no alignment.
#Classes
44,729
14,886
140,144
12,498
358,222
104,523
163,842
Ontology Pair
OMIM-ORDO</p>
        <p>NCIT-DOID
SNOMED-FMA
SNOMED-NCIT
SNOMED-NCIT</p>
        <p>Category
Disease
Disease
Body
Pharm
Neoplas</p>
        <p>We adopted a exible way of evaluating participating systems. First, participants can freely choose
any tasks and settings they would like to attend. Second, for systems that have been well-adapted to
the MELT platform, we used MELT to produce the output mappings. Third, for systems that have been
implemented elsewhere and are not easy to be made compatible with MELT, we used their source code.
Fourth, we also allowed participants (with trust) to directly upload output mappings if their systems
had not been published and had not been made compatible with MELT. In the nal result tables, we
used superscripts †, ‡, and * to indicate that the results came from MELT, source code implementation,
and direct result submission, respectively. All our evaluations were conducted with the DeepOnto18
[40] library.</p>
      </sec>
      <sec id="sec-3-6">
        <title>3.8. Biodiversity and Ecology</title>
        <p>The biodiversity and ecology (biodiv) track is motivated by the GFBio19 (The German Federation for
Biological Data) alongside its successor NFDI4Biodiversity20 and the AquaDiva21 projects, which aim at
providing semantically enriched data management solutions for data capture, annotation, indexing and
search [41, 42, 43]. In this track, we aim to motivate ontology matching systems to work on matching
ontologies and thesauri used in the biodiversity and ecology domains, available via the BiodivPortal
ontology repository22. For the current edition, we kept the matching task between the Environment
Ontology (ENVO) and the Semantic Web for Earth and Environment Technology Ontology (SWEET) as
these two ontologies have frequent updates.</p>
        <p>In 2021, we added a task to align two biological taxonomies with rather dierent but complementary
scopes: the well-known NCBI taxonomy (NCBITAXON), and TAXREF-LD [44]. No matching system
was able to achieve this matching task due to the large size of the considered taxonomies. To cope
with this issue since last year edition, we split the large matching task into a set of smaller, more
manageable subtasks through the use of modularization [45]. We obtained six groups corresponding to
the kingdoms: Animalia, Bacteria, Chromista, Fungi, Plantae and Protozoa, leading to six well balanced
matching subtasks.</p>
        <p>In 2023, we partnered with the EcoPortal project23 to include two new matching tasks involving
important thesauri in environmental sciences (originally developed in SKOS): nding alignments
between the Macroalgae Traits Thesaurus (MACROALGAE) and the Macrozoobenthos Traits Thesaurus
(MACROZOOBENTHOS) and between the Fish Traits Thesaurus (FISH) and the Zooplankton Traits
Thesaurus (ZOOPLANKTON). Table 4 presents detailed information about the ontologies and thesauri
used in this year’s edition.</p>
      </sec>
      <sec id="sec-3-7">
        <title>3.9. Digital Humanities</title>
        <p>The use of controlled vocabularies is widespread within the digital humanities (DH) [46]. The
development and usage of these vocabularies by dierent parties in related domains naturally leads to overlaps
in content [47]. While ontology matching helps with alignment and integration tasks, the application
of these systems to the digital humanities poses special challenges. Highly specic domain terminology
oen leads to smaller vocabularies, which oentimes include multiple (ancient) languages. Furthermore,
matching systems need to be compatible with SKOS vocabularies, since their use is fairly common
within the community.</p>
        <p>The DH track participated for the rst time this year. It includes eight test cases from archaeology,
cultural history and DH / computer science. Each test case consists of two SKOS (using RDF/XML as
16Created from OMIM texts by Mondo’s pipeline tool avaiable at: https://github.com/monarch-initiative/omim.
17Created by the ocial snomed-owl-toolkit available at: https://github.com/IHTSDO/snomed-owl-toolkit.
18https://krr-oxford.github.io/DeepOnto/#/
19www.gfbio.org
20www.nfdi4biodiversity.org/en/
21www.aquadiva.uni-jena.de
22biodivportal.gfbio.org/
23ecoportal.lifewatch.eu/
syntax) vocabularies to be matched and one manually created gold standard reference. For details on
the nine source vocabularies and on the test cases, see Table 5 and Table 6.</p>
        <p>The evaluation was executed on a virtual machine with 8 cores (2.4GHz each) and 16 GB RAM. To
quantify the performance, precision, recall and F1-score were used, while only evaluating equivalence
relationships. If matching systems resulted in either errors or zero identied matches, the task was
considered as failed. Adhering to the OAEI rules, no settings were changed before running the matching
systems.
24This is the eld to which the CV was grouped within our dataset.
25This is the number of concepts in the primary language of the CV before any preprocessing steps.
26https://vocabs.dariah.eu/defc thesaurus/en/
27https://isl.ics.forth.gr/bbt-federated-thesaurus/PACTOLS/en/
28https://vocabs.dariah.eu/iad thesaurus/en/
29https://isl.ics.forth.gr/bbt-federated-thesaurus/DAI/en/
30https://vocabs.dariah.eu/parthenos vocabularies/en/
31https://vocabs.acdh.oeaw.ac.at/oeai-cp/en/
32https://vocabs.dariah.eu/dha taxonomy/en/
33https://vocabularies.unesco.org/browser/thesaurus/en/
34https://vocabs.dariah.eu/tadirah/en/
35The number of terms varies depending on the branch used for the respective domain.</p>
      </sec>
      <sec id="sec-3-8">
        <title>3.10. Archaeology multilingual</title>
        <p>The archaeology multilingual track is based on an archaeology test case of the digital humanities track,
see section 3.9, with focus on evaluating matcher performance when dealing with dierent languages.</p>
        <p>Like the DH track, this track participated for the rst time. Each test case uses iDAI.world and
PACTOLS (see Table 5 for more information) as source resp. target. Both vocabularies contain terms in
English, French, German, and Italian. To create the ten test cases, all but one language were removed
from both vocabularies, leading to 10 dierent test cases, consisting of two monolingual vocabularies
and a manually created gold standard reference.</p>
        <p>The evaluation modalities are identical to ones in the digital humanities track, see section 3.9.</p>
      </sec>
      <sec id="sec-3-9">
        <title>3.11. Circular Economy</title>
        <p>In recent years, the Circular Economy (CE) domain has shown interest in representing domain knowledge
using ontologies. Since there are some existing CE-specic ontologies with more emerging, providing
alignments among ontologies can enhance the interoperability and reusability of such ontologies.
Therefore the circular economy track is proposed in 2024 consisting 1 task to match 2 ontologies in the
circular economy domain. These two ontologies are the Circular Economy Ontology Network [48] and
the Sustainable Bioeconomy and Bioproducts Ontology (BiOnto) [49]. CEON (including 214 classes)
from the Onto-DESIDE project,36 aims to represent core concepts for the CE domain [48]. BiOnto
(including 780 classes) from the BIOVOICES project,37 focuses on establishing a shared and common
terminology in the bioeconomy domain so that dierent stakeholders participating circular value
networks can provide information according to the ontology [50].</p>
        <p>The evaluation is conducted over standard parameters which are precision, recall, f-measure and
alignment size. The reference alignment for the matching task was initially done in [51] and further
validated by ontology engineers and CE domain experts from Onto-DESIDE project. The results is
presented in Section 4.12.</p>
      </sec>
      <sec id="sec-3-10">
        <title>3.12. Knowledge Graph</title>
        <p>The Knowledge Graph track was run for the fourth year. The task of the track is to match pairs of
knowledge graphs whose schema and instances have to be matched simultaneously. The individual
knowledge graphs are created by running the DBpedia extraction framework on eight dierent Wikis
from the Fandom Wiki hosting platform38 in the course of the DBkWik project [52, 53]. They cover
dierent topics (movies, games, comics, and books) and three Knowledge Graph clusters sharing the
same domain e.g., star trek, as shown in Table 7.</p>
        <p>The evaluation is based on reference correspondences at both schema and instance levels. While the
schema-level correspondences were created by experts, the instance correspondences were extracted
from the wiki page itself. Due to the fact that not all interwiki links on a page represent the same
36https://ontodeside.eu
37https://www.biovoices.eu
38https://www.wikia.com/
concept, a few restrictions were made: 1) only links in sections with a header containing “link” are used,
2) all links are removed where the source page links to more than one concept in another wiki (ensures
the alignments are functional), 3) multiple links which point to the same concept are also removed
(ensures injectivity), 4) links to disambiguation pages were manually checked and corrected. Since we
do not have a correspondence for each instance, class, and property in the graphs, this gold standard is
only a partial gold standard.</p>
        <p>The evaluation was executed on a virtual machine (VM) with 32GB of RAM and 16 vCPUs (2.4 GHz),
with Debian 9 operating system and Openjdk version 1.8.0 265. For evaluating all possible submission
formats, MELT framework is used. The corresponding code for evaluation can be found on Github39.</p>
        <p>The alignments were evaluated based on precision, recall, and f-measure for classes, properties, and
instances (each in isolation). The partial gold standard contained 1:1 correspondences, and we further
assume that in each knowledge graph, only one representation of the concept exists. This means that if
we have a correspondence in our gold standard, we count a correspondence to a dierent concept as a
false positive. The count of false negatives is only increased if we have a 1:1 correspondence and it is
not found by a matcher.</p>
        <p>As a baseline, we employed two simple string-matching approaches. The source code for these
matchers is publicly available40.</p>
      </sec>
      <sec id="sec-3-11">
        <title>3.13. Pharmacogenomics</title>
        <p>In 2024, the Pharmacogenomics track was run for the second time. This track focuses on matching
knowledge units from the pharmacogenomics domain. These units are -ary tuples – so-called
“pharmacogenomic relationships” – and involve drugs, genetic factors, and phenotypes (see Figure 1). A
pharmacogenomic tuple states that patients being treated by the specied drugs while having the
specied genetic factors may experience the given phenotypes.</p>
        <p>In the Semantic Web formalisms, only binary predicates exist. That is why pharmacogenomic tuples
are reied: tuples become individuals that are linked to their components with binary predicates
(Figure 1(c)). Hence, the task of matching pharmacogenomic tuples is [54]:
– An instance matching task that aims at nding alignments between individuals representing
reied tuples;
– A structure-based matching task in which neighbors of reied tuples are compared to conclude
the potential alignment between tuples. Recall that the only available information about these
tuples is their neighbors (e.g., no labels, or other properties).</p>
        <p>To illustrate, two tuples associating the same sets of drugs, genetic factors, and phenotypes have to the
same neighbors, thus represent the same two “pharmacogenomics relationships”, and thus should be
detected as identical.
39https://github.com/dwslab/melt/tree/master/examples/kgEvalCli
40http://oaei.ontologymatching.org/2019/results/knowledgegraph/kgBaselineMatchers.zip
{d1, . . . , d}
{gf1, . . . , gf}</p>
        <p>Beside the arity of tuples, matchers need to face issues such as incompleteness (e.g., missing drugs)
and heterogeneity (e.g., a gene version like CYP2C9*4 is more specic than the gene itself CYP2C9, the
phenotype hemorrhage is more specic than the phenotype vascular disorders). Dierent types
of alignments are thus expected to be identied between pharmacogenomic tuples, which is somehow
unusual in an instance matching task. The Pharmacogenomics track features the identication of
identical tuples (=), equivalent tuples (Close), tuples being more specic (&lt;) or more general (&gt;) than
others, and tuples being related to some extent (Related). See [54, 55] for a detailed denition of these
dierent alignment types between individuals.</p>
        <p>To perform this alignment task, matchers can rely on additional background knowledge about
components of pharmacogenomic tuples. This knowledge includes ontology classes instanciated by
the components of tuples (i.e. drugs, genetic factors, phenotypes) and their hierarchical organization,
partOf links between gene versions and genes, sameAs links between identical drugs, genes, or
phenotypes, and dependsOn links between complex phenotypes and their components (e.g.,
“warfarininduced bleeding” depends on “warfarin” and on “bleeding”).</p>
        <p>To evaluate matchers and their scalability, the Pharmacogenomics track comprises three tasks
involving respectively 10, 50, and 100% of the 50,435 pharmacogenomic tuples represented within the
PGxLOD knowledge graph41 [56]. For each task, the selected pharmacogenomic tuples are evenly split
into two ontologies to match. To take into account the specicity of the dierent alignment types that
are expected, matchers are evaluated through two settings:
Fine-grained setting Only alignments of the exact type expected in the reference are considered
correct. To illustrate, an output alignment (1, =, 2) where (1, Close, 2) was expected will be
considered as incorrect. Precision, Recall, and F1-score are computed for each type of alignment.
Coarse-grained setting Any type of alignment between entities expected to be aligned will be
considered as correct. To illustrate, an output alignment (1, =, 2) where (1, Close, 2) was expected
will be considered as correct. Precision, Recall, and F1-score are computed globally accordingly.</p>
      </sec>
    </sec>
    <sec id="sec-4">
      <title>4. Results and Discussion</title>
      <sec id="sec-4-1">
        <title>4.1. Participation</title>
        <p>Following an initial period of growth, the number of OAEI participants has remained approximately
constant since 2012, at slightly over 20. This year we count with 13 participating systems. Table 8 lists
the participants and the tracks in which they competed. It is worth mentioning that the Bio-ML track
has additional participants (e.g., BERTMap [57] and BERTSubs [58]) that are not counted in the number
of participants. This is because they need training and validation which are not yet fully supported
by the OAEI evaluation platforms, and thus they were tested locally with Bio-ML results reported,
0.931
0.908
0.881
0.848
0.75
0.728
0.622
0.36
and one matcher below StringEquiv baseline (TOMATO). Three matchers (ALIN, MDMapper, and
OntoMatch) do not match properties at all.</p>
        <p>The performance of all matching systems regarding their precision, recall and F1-measure is plotted
in Figure 2. Systems are represented as squares or triangles, whereas the baselines are represented as
circles.</p>
        <p>The Conference evaluation results using the uncertain reference alignments are presented in Table 11.
Out of the 7 alignment systems, 5 (ALIN, LogMapLt, MDMapper, OntoMatch, TOMATO) use 1.0 as the
condence value for all matches they identify. The remaining 2 systems (LogMap, Matcha) have a wide
variation of condence values.</p>
        <p>The evaluation results show key dierences in how matchers handle uncertain reference alignments,
particularly in discrete and continuous metrics.</p>
        <p>ALIN and OntoMatch both maintain high precision (0.88) across metrics and signicantly improve in
recall and F-measure when moving from sharp to uncertain settings, demonstrating strong adaptability
to uncertain alignments.</p>
        <p>LogMap and LogMapLt perform consistently, with LogMap maintaining stable precision (0.81) and
LogMapLt showing notable recall improvements from 0.50 in sharp to 0.63 in continuous. However, both
slightly drop in precision in uncertain metrics, reecting a cautious approach to condence assignments.</p>
        <p>Matcha and MDMapper adapt well to uncertain matches but face precision challenges. Matcha
sees high recall improvement from 0.67 in sharp to 0.77 in continuous, though with a precision dip in
uncertain metrics. MDMapper maintains stable recall across settings but loses precision, indicating
condence struggles with uncertain matches.</p>
        <p>F1-measure=0.7</p>
        <p>F1-measure=0.6
F1-measure=0.5</p>
        <p>TOMATO is the weakest performer, with limited recall and precision improvements, indicating
diculty capturing high-consensus matches condently.</p>
        <p>Overall, ALIN, LogMap, and OntoMatch excel with uncertain alignments, while TOMATO and
MDMapper struggle, highlighting the need for condence in uncertain data evaluation.</p>
      </sec>
      <sec id="sec-4-2">
        <title>4.4. Multifarm</title>
        <p>This year, 4 systems have registered to participate in the Multifarm track: LogMap, LogMapLt, Matcha
and MDMapper. The number of participating tools is similar with respect to the last 4 campaigns (4
in 2023, 5 in 2022, 6 in 2021, 6 in 2020, 5 in 2019). This year, we lost the participation of LSMatch
Multilingual. But we received new participation from MDMapper. The reader can refer to the OAEI
papers for a detailed description of the strategies adopted by each system.</p>
        <p>The Multifarm evaluation results based on the blind dataset are presented in Table 12, demonstrating
the aggregated results for the matching tasks. They have been computed using the MELT framework
without applying any threshold to the results. They are measured in terms of macro precision and recall.
The results of non-specic systems are not reported here, as we could observe in the last campaigns
that they can have intermediate results in tests of type ii) (same ontologies task) and poor performance
in tests i) (dierent ontologies task).</p>
        <p>The systems have been executed on a Windows Server 2025 machine congured with 96GB of RAM
running under a Intel Xeon Silver 4114 @2.20Ghz CPU, Tesla P40 GPU. All measurements are based on
a single run. As for each campaign, we observed large dierences in the time required for a system to
complete the 55 x 24 matching tasks:</p>
        <p>The results (Table 12) indicate notable dierences in performance across the four systems (LogMap,
LogMapLt, Matcha, and MDMapper) with regard to processing time, precision, F-measure, and recall.
LogMap exhibits the shortest processing time ( 13 minutes) and achieves the highest precision (0.72), but
its recall is relatively low (0.32), resulting in a moderate F-measure of 0.42. LogMapLt takes signicantly
longer ( 265 minutes) but shows much lower precision (0.24) and a minimal F-measure (0.038), along
with a low recall (0.02). Matcha requires even more time ( 309 minutes) and has a relatively balanced
performance, with a precision of 0.21, an F-measure of 0.28, and the highest recall among the systems
(0.44). Finally, MDMapper has the longest runtime ( 493 minutes) with low precision (0.25), recall (0.26),
and an F-measure of 0.04, indicating limited eectiveness despite the extended processing time. Overall,
LogMap stands out for its eciency and higher precision, while Matcha demonstrates better recall,
albeit at a signicant cost in processing time.</p>
      </sec>
      <sec id="sec-4-3">
        <title>4.5. Complex Matching</title>
        <p>Unfortunately, this track has not attracted many participants in the last years. This year only CANARD
has registered to participate. As CANARD depends on instances, it has only run on the Populated
Conference dataset. The system has been improved since its last participation in OAEI (2018), by
adopting embeddings generated by LLM. Table 13 shows the results of CANARD 2024 together with a
comparison to systems participating in the previous campaign (AMLC and Matcha-DL).</p>
        <p>Matcher
Matcha-DL
AMLC
CANARD 2018
CANARD 2024 (Stella-base IE 0.85)
CANARD 2024 (GritLM-7B ESQ)</p>
        <p>The results show that the integration of LLMs enhances the performance of CANARD, by increasing
the precision and F-measure by up to 45% over the baseline (CANARD 2018). These results corroborate
the eectiveness of such models in capturing semantic nuances. The congurations with Stella-base
model on the Instance Embeddings (IE) component and GritLM-7B model on the Embeddings of SPARQL
Query (ESQ) component of CANARD were the most eective.
4.6. Food
This is the third year of the track and three systems were registered: LogMap, LogMapLt and Matcha.</p>
        <p>The test case food v2 evaluates matching systems regarding their capability to nd “equal” (=) and
“subclass” relation (&lt;) correspondences between the CIQUAL ontology and the SIREN ontology. All
evaluated systems compute the alignment in less than a minute. LogMapLt stands out for its very fast
calculation time of 7s to nd “equal” (resp. “subclass” relation correspondences). Concerning “equal”
(=) relation correspondences, LogMap and LogMapLt have better precision than Matcha. However,
LogMap’s recall is 20 (resp. 11 times) less than Matcha’s one. Matcha is the best-performing participant
in the FNC test case in terms of precision and F1-measure. None of the matching systems are able to
nd “subclass” relation (&lt;) correspondences.</p>
      </sec>
      <sec id="sec-4-4">
        <title>4.7. Interactive matching</title>
        <p>This year, two systems (ALIN and LogMap) participated in the Interactive matching track. Their results
are shown in Table 15 and Figure 3 for both the Anatomy and Conference datasets.
NI stands for non-interactive, and refers to the results obtained by the matching system in the original track.</p>
        <p>The table includes the following information (column names within parentheses):
– The performance of the system: Precision (Prec.), Recall (Rec.), and F-measure (F-m.) with respect
to the xed reference alignment, as well as Recall+ (Rec.+) for the Anatomy task. To facilitate the
assessment of the impact of user interactions, we also provide the performance results from the
original tracks, without interaction (line with Error NI).
– To ascertain the impact of the oracle errors, we provide the performance of the system with
respect to the oracle (i.e., the reference alignment as modied by the errors introduced by the
oracle: Precision oracle (Prec. oracle), Recall oracle (Rec. oracle) and F-measure oracle (F-m.
oracle). For a perfect oracle, these values match the actual performance of the system.
– Total requests (Tot Reqs.) represents the number of distinct user interactions with the tool, where
each interaction can contain one to three conicting correspondences, that could be analyzed
simultaneously by a user.
– Distinct correspondences (Dist. Mapps) counts the total number of correspondences for which
the oracle gave feedback to the user (regardless of whether they were submitted simultaneously,
or separately).
– Finally, the performance of the oracle itself with respect to the errors it introduced can be gauged
through the positive precision (Pos. Prec.) and negative precision (Neg. Prec.), which measure
respectively the fraction of positive and negative answers given by the oracle that are correct.</p>
        <p>For a perfect oracle, these values are equal to 1 (or 0, if no questions were asked).</p>
        <p>The gure shows the time intervals between the questions to the user/oracle for the dierent systems
and error rates. Dierent runs are depicted with dierent colors.</p>
        <p>The matching systems that participated in this track employ dierent user-interaction strategies.
While LogMap makes use of user interactions exclusively in the post-matching steps to lter their
candidate correspondences, ALIN can also add new candidate correspondences to its initial set. LogMap
requests feedback on only selected correspondences candidates (based on their similarity patterns or
their involvement in unsatisabilities). ALIN and LogMap can both ask the oracle to analyze several
conicting correspondences simultaneously.</p>
        <p>The performance of the systems usually improves when interacting with a perfect oracle in
comparison with no interaction. ALIN is the system that improves the most, because of its high number of
oracle requests, and its non-interactive performance was the lowest of the interactive systems, and thus
the easiest to improve.</p>
        <p>Although system performance deteriorates when the error rate increases, there are still benets from
the user interaction—some of the systems’ measures stay above their non-interactive values even for
the larger error rates. Naturally, the more a system relies on the oracle, the more its performance tends
to be aected by the oracle’s errors.</p>
        <p>The impact of the oracle’s errors is linear for ALIN in most tasks, as the F-measure according to
the oracle remains approximately constant across all error rates. It is supra-linear for LogMap in all
datasets.</p>
        <p>Another aspect that was assessed, was the response time of systems, i.e., the time between requests.
Two models for system response times are frequently used in the literature [59]: Shneiderman and
Seow take dierent approaches to categorize the response times taking a task-centered view and a
user-centered view respectively. According to task complexity, Shneiderman denes response time in
four categories: typing, mouse movement (50-150 ms), simple frequent tasks (1 s), common tasks (2-4 s)
and complex tasks (8-12 s). While Seow’s denition of response time is based on the user expectations
towards the execution of a task: instantaneous (100-200 ms), immediate (0.5-1 s), continuous (2-5 s),
captive (7-10 s). Ontology alignment is a cognitively demanding task and can fall into the third or fourth
categories in both models. In this regard the response times (request intervals as we call them above)
observed in all datasets fall into the tolerable and acceptable response times, and even into the rst
categories, in both models. The request intervals for LogMap and ALIN stay at a few milliseconds for
most datasets. It could be the case, however, that a user would not be able to take advantage of these
low response times because the task complexity may result in higher user response time (i.e., the time
the user needs to respond to the system aer the system is ready).
4.8. Bio-ML
Our results include ve tables for equivalence matching, ve tables for subsumption matching, and
two tables for Bio-LLM, where each table corresponds to an OM pair and includes results of both the
unsupervised and semi-supervised settings. See Table 16 for an overview of the equivalence matching
results. For the full results, please refer to the OAEI 2024 Bio-ML website42.</p>
        <p>Briey, we have the following participants for equivalence matching: (i) machine learning-based
systems including BERTMap, BERTMapLt [57], BioGITOM, BioSTransMatch, HybridOM and Matcha
[60, 61]; and (ii) traditional systems including LogMap, LogMapBio, LogMapLt [39].</p>
        <p>In equivalence matching, the top-performing systems varied across tasks. BioGITOM achieved the
highest F1 score in 3 out of 5 semi-supervised tasks, while HybridOM led in the remaining two. For
unsupervised tasks, LogMapBio and HybridOM each attained the best F1 score in 2 out of 5 tasks, with
BERTMap excelling in the last one. Notably, BERTMap also achieved the best ranking scores on most
tasks, although some systems did not provide ranking results for equivalence matching. In subsumption
matching, no new systems participated this year.</p>
        <p>In summary the 2024 edition saw the introduction of three new machine learning-based systems.
While some participants from previous years did not resubmit their systems, the increased number
of machine learning-based participants aligns with Bio-ML’s original mission. Meanwhile, LogMap
variants remained the only symbolic systems in the competition.</p>
      </sec>
      <sec id="sec-4-5">
        <title>4.9. Biodiversity and Ecology</title>
        <p>This year, four matching systems (LogMap, LogMapLt, LogMapKG, and Matcha) managed to generate
an output for all of the track tasks, except Matcha failed to achieve alignment for the envo-sweet task.
As in previous editions, we used precision, recall, and F-measure to evaluate the performance of the
participating systems. The results for the Biodiversity and Ecology track are shown in Table 17.</p>
        <p>In comparison to the previous year, a smaller number of systems succeeded in generating alignments
for the track tasks. The results of the participating systems are comparable to last year in terms of
F-measure. In terms of run time, OLaLa took the longer. Regarding the ENVO-SWEET task, only OLaLa
and the LogMap family systems achieved it with a similar performance to last year. The
MACROALGAEMACROZOOBENTHOS and FISH-ZOOPLANKTON matching tasks involve resources developed in
SKOS. For the transformation, we made use of a source code directly derived from the AML ontology
parsing module, kindly provided to us by its developers. The systems that did not perform well in
this task did map a large number of dissimilar concepts that happen to have similar URIs. All systems
performed well on most NCBITAXON-TAXREF-LD subtasks, with slightly the same levels of precision
and recall. Overall, in this year’s evaluation, the number of participating systems decreased and the
performance of the successful ones remained similar.</p>
      </sec>
      <sec id="sec-4-6">
        <title>4.10. Digital Humanities</title>
        <p>Matcha, LogMap, LogMap Bio and LogMap KG found alignments. TOMATO and LogMap lite were
running without code errors, but resulted in empty alignments. ALIN and MDMapper had code
exceptions when executing. The same happened when trying to run OntoMatch, which was standalone
and not possible to run with MELT / SEALS.</p>
        <p>When comparing the matching systems (see table 18), LogMap KG has the best averaged F1-score
of 0.61. It is noteworthy that LogMap KG’s performance is more stable across tracks compared to the
second best, Matcha. The latter performed very well in some test cases, but poor in others.</p>
        <p>When we examine the F1-scores averaged over all matchers (see table 19), they range from 0.24 to
0.77. This indicates that while the matchers perform fairly well on some test cases, there is considerable
room for improvement on others.</p>
        <p>Looking at execution times (see table 20), they are all in the same range, between 13s and 21s to
run the full track. The only exception is LogMap lite with over 20 min but still results in an empty
alignment.</p>
        <p>In general, less than half of the evaluated matchers, and only one matcher that is not based on
LogMap, can nd alignments. Most of the systems resulted in errors, which aligns with our ndings
in our related OM-paper [62] where only ve out of 17 systems could nd alignments. This makes it
evident that most matching systems cannot handle SKOS even though SKOS is widely used in research
across dierent elds. This issue was already noted in the early library tracks [63] but has yet to be
addressed.</p>
      </sec>
      <sec id="sec-4-7">
        <title>4.11. Archaeology multilingual</title>
        <p>Since this track is composed of datasets of the digital humanities track, the matching systems that
found alignments resp. resulted in errors are identical in both tracks, therefore see section 4.10 for more
information.</p>
        <p>Comparing the matching systems (see table 21), LogMap Bio, and LogMap KG perform best with
an averaged F1-score of 0.26. This is in particular surprising for LogMap Bio because it was originally
developed / tuned for another domain.</p>
        <p>When looking at the F1-scores averaged over all matchers (see table 22), they range from 0.00 (nding
no alignments) to 0.59. Only the language combinations English-English and German-German are on
the upper end, while all the others are at or below 0.24. It is particularly interesting that for the language
combination French-Italian, not a single matching system was able to nd alignments. The results
suggest that most systems struggle when dealing with dierent languages. The fact that German and
English both belong to the West Germanic languages might be advantageous. The romance languages
pose a bigger challenge that the systems cannot solve in large parts.</p>
      </sec>
      <sec id="sec-4-8">
        <title>4.12. Circular Economy</title>
        <p>Three systems have been registered for the rst year of the Circular Economy track: LogMap, LogMapLt,
and Matcha. We conducted experiments by executing each system in its standard setting, and we
compared precision, F-measure, and recall. We used the MELT platform to execute our evaluations for
all systems.</p>
        <p>Table 24 shows the results for precision, F-measure, recall and the size of the alignments for the
optimal threshold. Regarding the recall, Matcha achieved the best score. LogMap and Matcha provide the
correspondences with real-valued condence. Therefore, we applied thresholding during the evaluation.</p>
        <p>The weights in LogMap’s alignment range from 0.93 to 0.14 (with only one weight below 0.5). The
mapping with the highest weight was a false positive, so was the mapping with the lowest weight.
There were multiple matches with the second lower weight (0.5). These mappings were a mix of correct
mappings and false positives. Based on these results, an optimal threshold for LogMap’s results could
be set to 0.5 (including) which is also the computed threshold with the highest F-measure.</p>
        <p>In case of Matcha, the weights of its results range between 1 and 0.600378464. Mappings with the
highest weights were both correct and false positives. The correct mapping with the lowest weight
was weighted to 0.65293388. This weight is just a little higher than the lowest weight. However, most
true positives have weights greater than 0.9. Based on these results, an optimal threshold for Matcha’s
results could be set to 0.9 which is also the computed threshold with the highest F-measure.</p>
        <p>Additionally, we analysed the false positives - alignments discovered by the tools which were
evaluated as incorrect. Looking at the results, it can be said that when the reason for discovering an
alignment was the same name, all or at least most tools generated the mapping. LogMap and Matcha
further generated mappings based on similar strings. All three systems generated mappings where the
same word was present in the entities’ names. Lastly, Matcha produced 2 mappings where the reason is
not obvious. As a possibly interesting observation, there were no false positives found which would be
generated based on synonyms in entities’ names. More information is provided at the results web page.</p>
        <p>The rst evaluation within the track shows that matching circular economy relevant ontologies
remains a challenging task for tools (F1-measure lower than 0.48). Based on false positives analysis, it
turns out that mere string matching could be misleading, and the meaning of entities should be better
considered.</p>
      </sec>
      <sec id="sec-4-9">
        <title>4.13. Knowledge Graph</title>
        <p>This year we evaluated all participants with the MELT framework to include all possible submission
formats i.e. SEALS, and Web format. First, all systems are evaluated on a very small matching task43
(even those not registered for the track). This revealed that not all systems were able to handle the task,
and in the end, 6 matchers can provide results for at least one test case.</p>
        <p>Table 25 shows the results for all systems divided into class, property, instance, and overall results.
This also includes the number of tasks in which they were able to generate a non-empty alignment
(#tasks) and the average number of generated correspondences (size). We report the macro averaged
precision, F-measure, and recall results, where we do not distinguish empty and erroneous (or not
generated) alignments. The values in parentheses show the results when considering only nonempty
alignments.</p>
        <p>The resulting alignments are available for download 44. This year’s best overall system is still the
baseline using the alternative labels (0.84 F-measure). The highest recall is again achieved by Matcha
(0.84). Detailed results for each test case can be found on the OAEI results page of the track45.</p>
        <p>Property matches are still not created by all systems. LogMap, Matcha, and MDMapper do not return
any of those mappings. One reason might be that the properties are typed as rdf:Property and not
distinguished into owl:ObjectProperty or owl:DatatypeProperty.</p>
        <p>When it comes to class matches, Matcha is the overall best system with an F-measure of 0.87 (much
better than the provided baseline).</p>
        <p>For further analysis of the results, we also provide an online dashboard46 generated with MELT[64].
In this dashboard, the results can be inspected on a correspondence level. Due to the large amount of
these correspondences, it can take some time to load the full website.</p>
        <p>Regarding runtime, Matcha (38:48:16) and LogMapLt (64:48:07) were the slowest systems. Besides the
baselines (which need around 12 minutes for all test cases) LogMap (00:56:43) is the fastest system.
43http://oaei.ontologymatching.org/2019/results/knowledgegraph/small test.zip
44http://oaei.ontologymatching.org/2024/results/knowledgegraph/knowledgegraph-alignments.zip
45http://oaei.ontologymatching.org/2024/results/knowledgegraph/index.html
46http://oaei.ontologymatching.org/2024/results/knowledgegraph/knowledge graph dashboard.html</p>
        <p>BaselineAltLabel 00:11:37 5
BaselineLabel 00:11:27 5
LogMap 00:56:43 5
LogMapLt 64:48:07 4
Matcha 38:48:16 5
MDMapper 02:28:53 5
BaselineAltLabel 00:11:37 5
BaselineLabel 00:11:27 5
LogMap 00:56:43 5
LogMapLt 64:48:07 4
Matcha 38:48:16 5
MDMapper 02:28:53 5
BaselineAltLabel 00:11:37 5
BaselineLabel 00:11:27 5
LogMap 00:56:43 5
LogMapLt 64:48:07 4
Matcha 38:48:16 5
MDMapper 02:28:53 5
BaselineAltLabel 00:11:37 5
BaselineLabel 00:11:27 5
LogMap 00:56:43 5
LogMapLt 64:48:07 4
Matcha 38:48:16 5
MDMapper 02:28:53 5</p>
      </sec>
      <sec id="sec-4-10">
        <title>4.14. Pharmacogenomics</title>
        <p>For this second year of the Pharmacogenomics track, only LogMap registered with its dierent versions:
LogMap, LogMap-Bio, LogMap-Lite, and LogMap-KG. The evaluation was performed using the MELT
framework.</p>
        <p>None of these versions were successful in producing alignments between reied -ary tuples. We
identify the main reason as the absence of labels for -ary tuples, since providing labels allows the
LogMap versions to produce alignments. However, when labels are present, altering neighborhoods
does not impact the produced alignments, showing that only labels are taken into account by the
dierent versions of the LogMap system. Recall that -ary tuples are reifed as abstract entities because
RDF does not allow -ary relations. Hence, labels of such reied entities are seldom present in general,
but their neighbors play a crucial role in their identity. This makes us conclude that submitted systems
are not adequate to the task of matching pharmacogenomic knowledge as they appear to rely only on
labels and disregard neighbors. It is also noteworthy that, without adding labels for -ary tuples, some
versions of LogMap output alignments between other entities (e.g., components of pharmacogenomic
tuples) but not between the -ary tuples themselves. Such alignments are valid but sometimes trivial
(e.g., between entities in the two ontologies to match that actually share the same URI) and out of the
scope of the Pharmacogenomics track.</p>
      </sec>
    </sec>
    <sec id="sec-5">
      <title>5. Conclusions and Lessons Learned</title>
      <p>As in previous campaigns, we witnessed a healthy mix of new and returning systems, with an imbalanced
participation in the tracks.</p>
      <p>The schema matching tracks gather the highest number of participants; however still little
substantial progress in terms of the quality of the results or run time of top matching systems. As already
reported in the last years, we observe a performance plateau being reached by existing strategies and
algorithms. It is also true that established matching systems tend to focus more on new tracks and
datasets than on improving their performance in long-standing tracks, whereas new systems typically
struggle to compete with established ones.</p>
      <p>With respect to the cross-lingual version of the Conference, the Multifarm track still attracts too few
number of participants. Despite this fact, this year new participants came up with alternative strategies
(i.e., deep learning) with respect to the last campaigns.</p>
      <p>In the Food track, none of the evaluated matchers nds all reference correspondences correctly.
LogMapLt stands out for its very fast computing speed. Matcha obtains the best results for the FNC
application. The usage of background knowledge available in CIQUAL and SIREN ontologies in terms
of food description based on FoodON concepts should be considered in future OAEI campaigns.</p>
      <p>The Bio-ML track incorporated signicant updates and attracted several new machine learning-based
participants. However, the number of symbolic participants decreased. The best-performing systems
are not consistent across tasks and settings, demonstrating the diversity of our datasets. It is also worth
noting that SORBETMatcher is the only system can participate in both equivalence and subsumption
matching.</p>
      <p>In the Biodiversity and Ecology track, none of the systems was able to detect manual mappings
created by domain experts and requiring biodiversity domain-specic knowledge. In this year’s edition,
we conrmed the inability of most systems to handle SKOS natively, as well as very large ontologies.
Additionally, some systems did not perform well on the thesauri tasks because those contained concepts
with similar URIs that were, in fact, completely dierent.</p>
      <p>The results of the Digital Humanities track clearly show that SKOS vocabularies are not
wellsupported by most matching systems. Regarding the matching systems that were able to nd alignments,
there is still room to improve, especially when the results are used for more complex alignment and
mapping tasks. To further improve this track, it is planned to include more subdomains of the digital
humanities.</p>
      <p>The Archaeology multilingual track leads to the conclusion that dierent languages within SKOS
vocabularies are not adequately supported, especially when coming to the family of Romance languages.
In future track versions, it is aimed for including additional ancient languages like Latin or Ancient
Greek.</p>
      <p>The Interactive matching track also witnessed a small number of participants. Two systems
participated this year. This is puzzling considering that this track is based on the Anatomy and
Conference test cases, and those tracks had 7 participants, respectively. The process of programmatically
querying the Oracle class used to simulate user interactions is simple enough that it should not be a
deterrent for participation, but perhaps we should look at facilitating the process further in future OAEI
editions by providing implementation examples.</p>
      <p>The Complex matching track tackles a challenge task that attracts too few number of participants.
This year, only one system was able to complete the task. As several sub-tracks have been discontinued,
the track is limited to the conference domain. This track welcomes new organizers.</p>
      <p>Automatic instance-matching benchmark generation algorithms have been gaining popularity, as
evidenced by the fact that they are used in all three instance-matching tracks of this OAEI edition. One
aspect that has not been addressed in such algorithms is that, if the transformation is too extreme,
the correspondence may be unrealistic and impossible to detect even by humans. As such, we argue
that human-in-the-loop techniques can be exploited to do a preventive quality-checking of generated
correspondences and rene the set of correspondences included in the nal reference alignment.</p>
      <p>In the Knowledge graph track, the overall best scores are still unbeaten. Furthermore, the proportion
of matchers not able to produce property alignments is high. This might change next year with new
and improved systems.</p>
      <p>For the second year of the Pharmacogenomics track, participation was limited with only four
versions of a single system registered, namely LogMap. None of these versions were successful in
producing alignments between reied -ary tuples, which, according to our investigation, is due to the
absence of labels for the tuples of our dataset. These results highlight again the interest in considering
domain-specic problems, bringing additional challenges to the eld of ontology matching (here,
dierent types of alignments between individuals, structure-based matching). Given the inadequacy of
registered systems to produce valid alignments, such challenges are currently unaddressed and require
to design new methods like [55, 65] or enrich existing ones. This ultimately motivates to propose again
the track in the next editions of OAEI, hoping to attract new systems targeting this real-world matching
scenario.</p>
      <p>Like in previous OAEI editions, most participants provided a description of their systems and their
experience in the evaluation, in the form of OAEI system papers. These papers, like the present one,
have not been peer-reviewed. However, they are full contributions to this evaluation exercise, reecting
the eort and insight of matching systems developers, and providing details about those systems and
the algorithms they implement.</p>
      <p>As each year, fruitful discussions at the Ontology Matching Workshop point out dierent directions
for future improvements in OAEI. This year, with a higher number of systems relying on Large Language
Models, there was a discussion on the specic requirements and alternative ways for gathering the
alignments generated by such resource-consuming systems. It has also been highlighted the need to
push the adoption of SSSOM [25] (since 2023 MELT has incorporated the format but still few systems
have adopted it), as a way for delivering richer alignments in terms of metadata and justications [66].
As already mentioned before, there were also some interrogations on the stability reached in some
(open)-schema matching tasks (in particular Anatomy and Conference tracks) as the performance has
been quite stable for several years. This requires a further analysis of the dicult parts of the matching
task. Last but not least, new tracks addressing more application/use-oriented tasks should be addressed
and they are more than welcome.</p>
      <p>The Ontology Alignment Evaluation Initiative will strive to remain a reference to the ontology
matching community by improving both the test cases and the testing methodology to better reect
actual needs, as well as to promote progress in this eld. More information can be found at: http:
//oaei.ontologymatching.org.</p>
    </sec>
    <sec id="sec-6">
      <title>Acknowledgments</title>
      <p>We warmly thank the participants of this campaign. We know that they have worked hard to have their
matching tools executable in time and they provided useful reports on their experience. The best way
to learn about the results remains to read the papers that follow.</p>
      <p>We are also grateful to Martin Ringwald and Terry Hayamizu for providing the reference alignment
for the anatomy ontologies and thank Elena Beisswanger for her thorough support in improving the
dataset’s quality.</p>
      <p>We also thank for their support, the past members of the Ontology Alignment Evaluation Initiative
steering committee: Je´ro^me Euzenat (INRIA, FR), Yannis Kalfoglou (Ricoh laboratories, UK), Miklos
Nagy (The Open University, UK), Natasha Noy (Google Inc., USA), Yuzhong Qu (Southeast University,
CN), York Sure (Leibniz Gemeinscha, DE), Jie Tang (Tsinghua University, CN), Heiner Stuckenschmidt
(Mannheim Universita¨t, DE), and George Vouros (University of the Aegean, GR).</p>
      <p>Daniel Faria and Catia Pesquita were supported by the FCT through the LASIGE Research Unit
(UIDB/00408/2020 and UIDP/00408/2020) and by the KATY project funded by the European Union’s
Horizon 2020 research and innovation program under grant agreement No 101017453.</p>
      <p>Patrick Lambrix, Huanyu Li, Mina Abd Nikooie Pour and Ying Li have been supported by the Swedish
e-Science Research Centre (SeRC) and the Swedish National Graduate School in Computer Science
(CUGS).</p>
      <p>Eva Blomqvist, Patrick Lambrix, Huanyu Li, Ondrˇej Zamazal and Jana Vatasˇcˇinova´ have been
supported by the European Union’s Horizon Europe research and innovation programme under grant
agreement no. 101058682 (Onto-DESIDE).</p>
      <p>Beyza Yaman has been supported by ADAPT SFI Research Centre [grant 13/RC/2106 P2].</p>
      <p>Jiaoyan Chen, Hang Dong, Yuan He, and Ian Horrocks have been supported by Samsung Research
UK (SRUK) and the EPSRC project ConCur (EP/V050869/1).</p>
      <p>Naouel Karam and Alsayed Algergawy have been supported by the German Research Foundation in
the context of NFDI4BioDiversity project (number 442032008) and the CRC 1076 AquaDiva. We would
like to thank Jessica Titocci, Martina Pulieri and Ilaria Rosati for providing the datasets for the biodiv
SKOS thesauri tasks.</p>
      <p>The work of Felix Kraus was funded by the research program “Engineering Digital Futures” of
the Helmholtz Association of German Research Centers, and the Helmholtz Metadata Collaboration
Platform (HMC).
3233/SW-210437.
[30] P. Buche, J. Cu, S. Dervaux, J. Dibie, L. Ibanescu, A. Oudot, M. Weber, How to manage
incompleteness of nutritional food sources?: A solution using foodon as pivot ontology, Int.
J. Agric. Environ. Inf. Syst. 12 (2021) 1–26. URL: https://doi.org/10.4018/ijaeis.20211001.oa4.
doi:10.4018/ijaeis.20211001.oa4.
[31] H. Paulheim, S. Hertling, D. Ritze, Towards evaluating interactive ontology matching tools, in:
Proceedings of the 10th Extended Semantic Web Conference, Montpellier (FR), 2013, pp. 31–45.</p>
      <p>URL: http://dx.doi.org/10.1007/978-3-642-38288-8 3.
[32] Z. Dragisic, V. Ivanova, P. Lambrix, D. Faria, E. Jime´nez-Ruiz, C. Pesquita, User validation
in ontology alignment, in: Proceedings of the 15th International Semantic Web Conference,
Kobe (JP), 2016, pp. 200–217. URL: http://dx.doi.org/10.1007/978-3-319-46523-4 13. doi:10.1007/
978-3-319-46523-4 13.
[33] H. Li, Z. Dragisic, D. Faria, V. Ivanova, E. Jime´nez-Ruiz, P. Lambrix, C. Pesquita, User validation in
ontology alignment: functional assessment and impact, The Knowledge Engineering Review 34
(2019) e15. doi:10.1017/S0269888919000080.
[34] V. Ivanova, P. Lambrix, J. ˚Aberg, Requirements for and evaluation of user support for large-scale
ontology alignment, in: Proceedings of the European Semantic Web Conference, 2015, pp. 3–20.
[35] Y. He, J. Chen, H. Dong, E. Jime´nez-Ruiz, A. Hadian, I. Horrocks, Machine learning-friendly
biomedical datasets for equivalence and subsumption ontology matching, in: U. Sattler, A. Hogan,
C. M. Keet, V. Presutti, J. P. A. Almeida, H. Takeda, P. Monnin, G. Pirro`, C. d’Amato (Eds.), The
Semantic Web - ISWC 2022 - 21st International Semantic Web Conference, Virtual Event, October
23-27, 2022, Proceedings, volume 13489 of Lecture Notes in Computer Science, Springer, 2022, pp. 575–
591. URL: https://doi.org/10.1007/978-3-031-19433-7 33. doi:10.1007/978-3-031-19433-7∖
33.
[36] N. A. Vasilevsky, N. A. Matentzoglu, S. Toro, J. E. Flack IV, H. Hegde, D. R. Unni, G. F. Alyea, J. S.</p>
      <p>Amberger, L. Babb, J. P. Balho, et al., Mondo: Unifying diseases for the world, by the world,
medRxiv (2022) 2022–04.
[37] O. Bodenreider, The unied medical language system (umls): integrating biomedical terminology,</p>
      <p>Nucleic acids research (2004).
[38] Y. He, J. Chen, H. Dong, I. Horrocks, Exploring large language models for ontology alignment,
arXiv preprint arXiv:2309.07172 (2023).
[39] E. Jime´nez-Ruiz, B. C. Grau, LogMap: Logic-based and scalable ontology matching, in: Proceedings
of the 10th International Semantic Web Conference, Bonn (DE), 2011, pp. 273–288.
[40] Y. He, J. Chen, H. Dong, I. Horrocks, C. Allocca, T. Kim, B. Sapkota, Deeponto: A python package
for ontology engineering with deep learning, arXiv preprint arXiv:2307.03067 (2023).
[41] N. Karam, C. Mu¨ller-Birn, M. Gleisberg, D. Fichtmu¨ller, R. Tolksdorf, A. Gu¨ntsch, A
terminology service supporting semantic annotation, integration, discovery and analysis of
interdisciplinary research data, Datenbank-Spektrum 16 (2016) 195–205. URL: https://doi.org/10.1007/
s13222-016-0231-8. doi:10.1007/s13222-016-0231-8.
[42] F. Klan, E. Faessler, A. Algergawy, B. Ko¨nig-Ries, U. Hahn, Integrated semantic search on structured
and unstructured data in the adonis system, in: Proceedings of the 2nd International Workshop
on Semantics for Biodiversity, 2017.
[43] N. Karam, A. Khiat, A. Algergawy, M. Sattler, C. Weiland, M. Schmidt, Matching biodiversity
and ecology ontologies: challenges and evaluation results, Knowl. Eng. Rev. 35 (2020) e9. URL:
https://doi.org/10.1017/S0269888920000132. doi:10.1017/S0269888920000132.
[44] F. Michel, O. Gargominy, S. Tercerie, C. Faron-Zucker, A Model to Represent Nomenclatural and
Taxonomic Information as Linked Data. Application to the French Taxonomic Register, TAXREF, in:
A. Algergawy, N. Karam, F. Klan, C. Jonquet (Eds.), Proceedings of the 2nd International Workshop
on Semantics for Biodiversity co-located with 16th International Semantic Web Conference (ISWC
2017), Vienna, Austria, October 22nd, 2017, volume 1933 of CEUR Workshop Proceedings,
CEURWS.org, 2017.
[45] A. Algergawy, N. Karam, A. Laadhar, F. Michel, Too big to match: a strategy around matching tasks
2022., in: OM@ ISWC, 2022, pp. 197–201.
[61] D. Faria, M. C. Silva, P. Cotovio, L. Ferraz, L. Balbi, C. Pesquita, Results for matcha and matcha-dl
in oaei 2023., in: OM@ ISWC, 2023, pp. 164–169.
[62] F. Kraus, N. Blumenro¨hr, G. Go¨tzelmann, D. Tonne, A. Streit, A Gold Standard Benchmark Dataset
for Digital Humanities, in: Proceedings of the 19th International Workshop on Ontology Matching,
CEUR Workshop Proceedings, Baltimore, USA, in press.
[63] C. Caracciolo, J. Euzenat, L. Hollink, R. Ichise, A. Isaac, V. Malaise´, C. Meilicke, J. Pane, P. Shvaiko,
H. Stuckenschmidt, O. Sva´b-Zamazal, V. Sva´tek, Results of the Ontology Alignment Evaluation
Initiative 2008, in: P. Shvaiko, J. Euzenat, F. Giunchiglia, H. Stuckenschmidt (Eds.), Proceedings of
the 3rd International Workshop on Ontology Matching (OM-2008), volume 431 of CEUR Workshop
Proceedings, CEUR-WS.org, Karlsruhe, Germany, 2008.
[64] J. Portisch, S. Hertling, H. Paulheim, Visual analysis of ontology matching results with the melt
dashboard, in: European Semantic Web Conference, 2020, pp. 186–190.
[65] P. Monnin, C. Ra¨ıssi, A. Napoli, A. Coulet, Discovering alignment relations with graph
convolutional networks: A biomedical case study, Semantic Web 13 (2022) 379–398. URL: https:
//doi.org/10.3233/SW-210452. doi:10.3233/SW-210452.
[66] N. Matentzoglu, I. Braun, A. R. Caron, D. Goutte-Gattat, B. M. Gyori, N. L. Harris, E. Hartley, H. B.</p>
      <p>Hegde, S. Hertling, C. Tapley, H. Kim, H. Li, J. McLaughlin, C. Trojahn, N. Vasilevsky, C. Mungall,
A Simple Standard for Ontological Mappings 2023: Updates on data model, collaborations and
tooling, in: Proceedings of the 18th International Workshop on Ontology Matching (OM 2023)
co-located with the 22nd International Semantic Web Conference (ISWC 2023), Athens, Greece,
November 7, 2023, volume 3591 of CEUR Workshop Proceedings, CEUR-WS.org, 2023. URL: https:
//ceur-ws.org/Vol-3591/om2023 STpaper3.pdf.</p>
    </sec>
  </body>
  <back>
    <ref-list>
      <ref id="ref1">
        <mixed-citation>
          [1]
          <string-name>
            <given-names>J.</given-names>
            <surname>Euzenat</surname>
          </string-name>
          ,
          <string-name>
            <given-names>C.</given-names>
            <surname>Meilicke</surname>
          </string-name>
          ,
          <string-name>
            <given-names>P.</given-names>
            <surname>Shvaiko</surname>
          </string-name>
          ,
          <string-name>
            <given-names>H.</given-names>
            <surname>Stuckenschmidt</surname>
          </string-name>
          ,
          <string-name>
            <surname>C.</surname>
          </string-name>
          <article-title>Trojahn dos Santos, Ontology alignment evaluation initiative: six years of experience</article-title>
          ,
          <source>Journal on Data Semantics XV</source>
          (
          <year>2011</year>
          )
          <fpage>158</fpage>
          -
          <lpage>192</lpage>
          .
        </mixed-citation>
      </ref>
      <ref id="ref2">
        <mixed-citation>
          [2]
          <string-name>
            <given-names>J.</given-names>
            <surname>Euzenat</surname>
          </string-name>
          ,
          <string-name>
            <given-names>P.</given-names>
            <surname>Shvaiko</surname>
          </string-name>
          , Ontology matching, 2nd ed., Springer-Verlag,
          <year>2013</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref3">
        <mixed-citation>
          [3]
          <string-name>
            <given-names>Y.</given-names>
            <surname>Sure</surname>
          </string-name>
          ,
          <string-name>
            <given-names>O.</given-names>
            <surname>Corcho</surname>
          </string-name>
          ,
          <string-name>
            <given-names>J.</given-names>
            <surname>Euzenat</surname>
          </string-name>
          , T. Hughes (Eds.),
          <source>Proceedings of the Workshop on Evaluation of Ontology-based Tools (EON)</source>
          ,
          <source>Hiroshima (JP)</source>
          ,
          <year>2004</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref4">
        <mixed-citation>
          [4]
          <string-name>
            <given-names>B.</given-names>
            <surname>Ashpole</surname>
          </string-name>
          ,
          <string-name>
            <given-names>M.</given-names>
            <surname>Ehrig</surname>
          </string-name>
          ,
          <string-name>
            <given-names>J.</given-names>
            <surname>Euzenat</surname>
          </string-name>
          , H. Stuckenschmidt (Eds.),
          <string-name>
            <surname>Proc.</surname>
          </string-name>
          K-Cap Workshop on Integrating Ontologies, Ban (Canada),
          <year>2005</year>
          . URL: http://ceur-ws.
          <source>org/</source>
          Vol-
          <volume>156</volume>
          /.
        </mixed-citation>
      </ref>
      <ref id="ref5">
        <mixed-citation>
          [5]
          <string-name>
            <given-names>M.</given-names>
            <surname>Abd Nikooie Pour</surname>
          </string-name>
          ,
          <string-name>
            <given-names>A.</given-names>
            <surname>Algergawy</surname>
          </string-name>
          ,
          <string-name>
            <given-names>P.</given-names>
            <surname>Buche</surname>
          </string-name>
          ,
          <string-name>
            <given-names>L. J.</given-names>
            <surname>Castro</surname>
          </string-name>
          ,
          <string-name>
            <given-names>J.</given-names>
            <surname>Chen</surname>
          </string-name>
          ,
          <string-name>
            <given-names>A.</given-names>
            <surname>Coulet</surname>
          </string-name>
          , J. Cu,
          <string-name>
            <given-names>H.</given-names>
            <surname>Dong</surname>
          </string-name>
          ,
          <string-name>
            <given-names>O.</given-names>
            <surname>Fallatah</surname>
          </string-name>
          ,
          <string-name>
            <given-names>D.</given-names>
            <surname>Faria</surname>
          </string-name>
          , I. Fundulaki,
          <string-name>
            <given-names>S.</given-names>
            <surname>Hertling</surname>
          </string-name>
          ,
          <string-name>
            <given-names>Y.</given-names>
            <surname>He</surname>
          </string-name>
          ,
          <string-name>
            <surname>I. Horrocks</surname>
          </string-name>
          ,
          <string-name>
            <given-names>M.</given-names>
            <surname>Huschka</surname>
          </string-name>
          ,
          <string-name>
            <given-names>L.</given-names>
            <surname>Ibanescu</surname>
          </string-name>
          ,
          <string-name>
            <given-names>S.</given-names>
            <surname>Jain</surname>
          </string-name>
          ,
          <string-name>
            <surname>E.</surname>
          </string-name>
          <article-title>Jime´nez-</article-title>
          <string-name>
            <surname>Ruiz</surname>
            ,
            <given-names>N.</given-names>
          </string-name>
          <string-name>
            <surname>Karam</surname>
            ,
            <given-names>P.</given-names>
          </string-name>
          <string-name>
            <surname>Lambrix</surname>
            ,
            <given-names>H.</given-names>
          </string-name>
          <string-name>
            <surname>Li</surname>
            ,
            <given-names>Y.</given-names>
          </string-name>
          <string-name>
            <surname>Li</surname>
            ,
            <given-names>P.</given-names>
          </string-name>
          <string-name>
            <surname>Monnin</surname>
            , E. Nasr,
            <given-names>H.</given-names>
          </string-name>
          <string-name>
            <surname>Paulheim</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          <string-name>
            <surname>Pesquita</surname>
            ,
            <given-names>T.</given-names>
          </string-name>
          <string-name>
            <surname>Saveta</surname>
            ,
            <given-names>P.</given-names>
          </string-name>
          <string-name>
            <surname>Shvaiko</surname>
            , G. Sousa,
            <given-names>C.</given-names>
          </string-name>
          <string-name>
            <surname>Trojahn</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          <string-name>
            <surname>Vatascinova</surname>
            ,
            <given-names>M.</given-names>
          </string-name>
          <string-name>
            <surname>Wu</surname>
            ,
            <given-names>B.</given-names>
          </string-name>
          <string-name>
            <surname>Yaman</surname>
            ,
            <given-names>O.</given-names>
          </string-name>
          <string-name>
            <surname>Zamazal</surname>
            ,
            <given-names>L.</given-names>
          </string-name>
          <string-name>
            <surname>Zhou</surname>
          </string-name>
          ,
          <article-title>Results of the ontology alignment evaluation initiative 2023</article-title>
          , in: P. Shvaiko,
          <string-name>
            <given-names>J.</given-names>
            <surname>Euzenat</surname>
          </string-name>
          , E. Jime´nezRuiz,
          <string-name>
            <given-names>O.</given-names>
            <surname>Hassanzadeh</surname>
          </string-name>
          ,
          <string-name>
            <surname>C.</surname>
          </string-name>
          Trojahn (Eds.),
          <source>Proceedings of the 18th International Workshop on Ontology Matching (OM</source>
          <year>2023</year>
          )
          <article-title>co-located with the 22nd International Semantic Web Conference (ISWC</article-title>
          <year>2023</year>
          ), Athens, Greece, November 7,
          <year>2023</year>
          , volume
          <volume>3591</volume>
          <source>of CEUR Workshop Proceedings, CEUR-WS.org</source>
          ,
          <year>2023</year>
          , pp.
          <fpage>97</fpage>
          -
          <lpage>139</lpage>
          . URL: https://ceur-ws.
          <source>org/</source>
          Vol-
          <volume>3591</volume>
          /oaei23 paper0.pdf.
        </mixed-citation>
      </ref>
      <ref id="ref6">
        <mixed-citation>
          [6]
          <string-name>
            <given-names>M.</given-names>
            <surname>Abd Nikooie Pour</surname>
          </string-name>
          ,
          <string-name>
            <given-names>A.</given-names>
            <surname>Algergawy</surname>
          </string-name>
          ,
          <string-name>
            <given-names>P.</given-names>
            <surname>Buche</surname>
          </string-name>
          ,
          <string-name>
            <given-names>L. J.</given-names>
            <surname>Castro</surname>
          </string-name>
          ,
          <string-name>
            <given-names>J.</given-names>
            <surname>Chen</surname>
          </string-name>
          ,
          <string-name>
            <given-names>H.</given-names>
            <surname>Dong</surname>
          </string-name>
          ,
          <string-name>
            <given-names>O.</given-names>
            <surname>Fallatah</surname>
          </string-name>
          ,
          <string-name>
            <given-names>D.</given-names>
            <surname>Faria</surname>
          </string-name>
          , I. Fundulaki,
          <string-name>
            <given-names>S.</given-names>
            <surname>Hertling</surname>
          </string-name>
          ,
          <string-name>
            <given-names>Y.</given-names>
            <surname>He</surname>
          </string-name>
          ,
          <string-name>
            <surname>I. Horrocks</surname>
          </string-name>
          ,
          <string-name>
            <given-names>M.</given-names>
            <surname>Huschka</surname>
          </string-name>
          ,
          <string-name>
            <given-names>L.</given-names>
            <surname>Ibanescu</surname>
          </string-name>
          ,
          <string-name>
            <surname>E.</surname>
          </string-name>
          <article-title>Jime´nez-</article-title>
          <string-name>
            <surname>Ruiz</surname>
            ,
            <given-names>N.</given-names>
          </string-name>
          <string-name>
            <surname>Karam</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          <string-name>
            <surname>Laadhar</surname>
            ,
            <given-names>P.</given-names>
          </string-name>
          <string-name>
            <surname>Lambrix</surname>
            ,
            <given-names>H.</given-names>
          </string-name>
          <string-name>
            <surname>Li</surname>
            ,
            <given-names>Y.</given-names>
          </string-name>
          <string-name>
            <surname>Li</surname>
            ,
            <given-names>F.</given-names>
          </string-name>
          <string-name>
            <surname>Michel</surname>
            , E. Nasr,
            <given-names>H.</given-names>
          </string-name>
          <string-name>
            <surname>Paulheim</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          <string-name>
            <surname>Pesquita</surname>
            ,
            <given-names>T.</given-names>
          </string-name>
          <string-name>
            <surname>Saveta</surname>
            ,
            <given-names>P.</given-names>
          </string-name>
          <string-name>
            <surname>Shvaiko</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          <string-name>
            <surname>Trojahn</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          <string-name>
            <surname>Verhey</surname>
            ,
            <given-names>M.</given-names>
          </string-name>
          <string-name>
            <surname>Wu</surname>
            ,
            <given-names>B.</given-names>
          </string-name>
          <string-name>
            <surname>Yaman</surname>
            ,
            <given-names>O.</given-names>
          </string-name>
          <string-name>
            <surname>Zamazal</surname>
            ,
            <given-names>L.</given-names>
          </string-name>
          <string-name>
            <surname>Zhou</surname>
          </string-name>
          ,
          <article-title>Results of the ontology alignment evaluation initiative 2022</article-title>
          , in: P. Shvaiko,
          <string-name>
            <given-names>J.</given-names>
            <surname>Euzenat</surname>
          </string-name>
          ,
          <string-name>
            <surname>E.</surname>
          </string-name>
          <article-title>Jime´nez-</article-title>
          <string-name>
            <surname>Ruiz</surname>
            ,
            <given-names>O.</given-names>
          </string-name>
          <string-name>
            <surname>Hassanzadeh</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          Trojahn (Eds.),
          <source>Proceedings of the 17th International Workshop on Ontology Matching (OM</source>
          <year>2022</year>
          )
          <article-title>co-located with the 21th International Semantic Web Conference (ISWC 2022), Hangzhou, China, held as a virtual conference</article-title>
          ,
          <source>October</source>
          <volume>23</volume>
          ,
          <year>2022</year>
          , volume
          <volume>3324</volume>
          <source>of CEUR Workshop Proceedings, CEUR-WS.org</source>
          ,
          <year>2022</year>
          , pp.
          <fpage>84</fpage>
          -
          <lpage>128</lpage>
          . URL: https://ceur-ws.
          <source>org/</source>
          Vol-
          <volume>3324</volume>
          /oaei22 paper0.pdf.
        </mixed-citation>
      </ref>
      <ref id="ref7">
        <mixed-citation>
          [7]
          <string-name>
            <given-names>M.</given-names>
            <surname>Abd Nikooie Pour</surname>
          </string-name>
          ,
          <string-name>
            <given-names>A.</given-names>
            <surname>Algergawy</surname>
          </string-name>
          ,
          <string-name>
            <given-names>F.</given-names>
            <surname>Amardeilh</surname>
          </string-name>
          ,
          <string-name>
            <given-names>R.</given-names>
            <surname>Amini</surname>
          </string-name>
          ,
          <string-name>
            <given-names>O.</given-names>
            <surname>Fallatah</surname>
          </string-name>
          ,
          <string-name>
            <given-names>D.</given-names>
            <surname>Faria</surname>
          </string-name>
          ,
          <string-name>
            <surname>I. Fundulaki</surname>
          </string-name>
          , I. Harrow,
          <string-name>
            <given-names>S.</given-names>
            <surname>Hertling</surname>
          </string-name>
          ,
          <string-name>
            <given-names>P.</given-names>
            <surname>Hitzler</surname>
          </string-name>
          ,
          <string-name>
            <given-names>M.</given-names>
            <surname>Huschka</surname>
          </string-name>
          ,
          <string-name>
            <given-names>L.</given-names>
            <surname>Ibanescu</surname>
          </string-name>
          ,
          <string-name>
            <surname>E.</surname>
          </string-name>
          <article-title>Jime´nez-</article-title>
          <string-name>
            <surname>Ruiz</surname>
            ,
            <given-names>N.</given-names>
          </string-name>
          <string-name>
            <surname>Karam</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          <string-name>
            <surname>Laadhar</surname>
            ,
            <given-names>P.</given-names>
          </string-name>
          <string-name>
            <surname>Lambrix</surname>
            ,
            <given-names>H.</given-names>
          </string-name>
          <string-name>
            <surname>Li</surname>
            ,
            <given-names>Y.</given-names>
          </string-name>
          <string-name>
            <surname>Li</surname>
            ,
            <given-names>F.</given-names>
          </string-name>
          <string-name>
            <surname>Michel</surname>
            , E. Nasr,
            <given-names>H.</given-names>
          </string-name>
          <string-name>
            <surname>Paulheim</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          <string-name>
            <surname>Pesquita</surname>
            ,
            <given-names>J.</given-names>
          </string-name>
          <string-name>
            <surname>Portisch</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          <string-name>
            <surname>Roussey</surname>
            ,
            <given-names>T.</given-names>
          </string-name>
          <string-name>
            <surname>Saveta</surname>
            ,
            <given-names>P.</given-names>
          </string-name>
          <string-name>
            <surname>Shvaiko</surname>
            ,
            <given-names>A.</given-names>
          </string-name>
          <string-name>
            <surname>Splendiani</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          <string-name>
            <surname>Trojahn</surname>
            , J. Vatascinova´,
            <given-names>B.</given-names>
          </string-name>
          <string-name>
            <surname>Yaman</surname>
            ,
            <given-names>O.</given-names>
          </string-name>
          <string-name>
            <surname>Zamazal</surname>
            ,
            <given-names>L.</given-names>
          </string-name>
          <string-name>
            <surname>Zhou</surname>
          </string-name>
          ,
          <article-title>Results of the ontology alignment evaluation initiative 2021</article-title>
          , in: P. Shvaiko,
          <string-name>
            <given-names>J.</given-names>
            <surname>Euzenat</surname>
          </string-name>
          ,
          <string-name>
            <surname>E.</surname>
          </string-name>
          <article-title>Jime´nez-</article-title>
          <string-name>
            <surname>Ruiz</surname>
            ,
            <given-names>O.</given-names>
          </string-name>
          <string-name>
            <surname>Hassanzadeh</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          Trojahn (Eds.),
          <source>Proceedings of the 16th International Workshop on Ontology Matching co-located with the 20th International Semantic Web Conference (ISWC</source>
          <year>2021</year>
          ), Virtual conference,
          <source>October</source>
          <volume>25</volume>
          ,
          <year>2021</year>
          , volume
          <volume>3063</volume>
          <source>of CEUR Workshop Proceedings, CEUR-WS.org</source>
          ,
          <year>2021</year>
          , pp.
          <fpage>62</fpage>
          -
          <lpage>108</lpage>
          . URL: http://ceur-ws.
          <source>org/</source>
          Vol-
          <volume>3063</volume>
          /oaei21 paper0.pdf.
        </mixed-citation>
      </ref>
    </ref-list>
  </back>
</article>