<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD v1.0 20120330//EN" "JATS-archivearticle1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink">
  <front>
    <journal-meta>
      <journal-title-group>
        <journal-title>Benchmark from Real-world Ontology, Data Intell.</journal-title>
      </journal-title-group>
    </journal-meta>
    <article-meta>
      <article-id pub-id-type="doi">10.1007/978-3-030-33220-4</article-id>
      <title-group>
        <article-title>Results of the Ontology Alignment Evaluation Initiative 2025</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="author">
          <string-name>Mina Abd Nikooie Pour</string-name>
          <xref ref-type="aff" rid="aff13">13</xref>
          <xref ref-type="aff" rid="aff5">5</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Eva Blomqvist</string-name>
          <xref ref-type="aff" rid="aff5">5</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Pedro Giesteira Cotovio</string-name>
          <xref ref-type="aff" rid="aff3">3</xref>
          <xref ref-type="aff" rid="aff9">9</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Adrien Coulet</string-name>
          <xref ref-type="aff" rid="aff2">2</xref>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Lucas Ferraz</string-name>
          <xref ref-type="aff" rid="aff9">9</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Sven Hertling</string-name>
          <xref ref-type="aff" rid="aff4">4</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Sarika Jain</string-name>
          <xref ref-type="aff" rid="aff10">10</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Ernesto Jime´nez-Ruiz</string-name>
          <xref ref-type="aff" rid="aff3">3</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Felix Kraus</string-name>
          <xref ref-type="aff" rid="aff8">8</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Patrick Lambrix</string-name>
          <xref ref-type="aff" rid="aff13">13</xref>
          <xref ref-type="aff" rid="aff5">5</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Huanyu Li</string-name>
          <xref ref-type="aff" rid="aff5">5</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Ying Li</string-name>
          <xref ref-type="aff" rid="aff13">13</xref>
          <xref ref-type="aff" rid="aff5">5</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Xianhao Liu</string-name>
          <xref ref-type="aff" rid="aff12">12</xref>
          <xref ref-type="aff" rid="aff14">14</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Pierre Monnin</string-name>
          <xref ref-type="aff" rid="aff17">17</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Heiko Paulheim</string-name>
          <xref ref-type="aff" rid="aff4">4</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Catia Pesquita</string-name>
          <xref ref-type="aff" rid="aff9">9</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Abhisek Sharma</string-name>
          <xref ref-type="aff" rid="aff10">10</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Pavel Shvaiko</string-name>
          <xref ref-type="aff" rid="aff15">15</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Marta Silva</string-name>
          <xref ref-type="aff" rid="aff9">9</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Guilherme Sousa</string-name>
          <xref ref-type="aff" rid="aff7">7</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Cassia Trojahn</string-name>
          <xref ref-type="aff" rid="aff16">16</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Jana Vatasˇcˇinova´</string-name>
          <xref ref-type="aff" rid="aff11">11</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Beyza Yaman</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Ondrˇej Zamazal</string-name>
          <xref ref-type="aff" rid="aff11">11</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Lu Zhou</string-name>
          <xref ref-type="aff" rid="aff1">1</xref>
        </contrib>
        <aff id="aff0">
          <label>0</label>
          <institution>ADAPT Centre, Trinity College Dublin</institution>
        </aff>
        <aff id="aff1">
          <label>1</label>
          <institution>Apple Inc.</institution>
          ,
          <country country="US">USA</country>
        </aff>
        <aff id="aff2">
          <label>2</label>
          <institution>Centre de Recherche des Cordeliers, Inserm, Universite ́ Paris Cite ́, Sorbonne Universite ́</institution>
          ,
          <country country="FR">France</country>
        </aff>
        <aff id="aff3">
          <label>3</label>
          <institution>City St George's, University of London</institution>
          ,
          <country country="UK">UK</country>
        </aff>
        <aff id="aff4">
          <label>4</label>
          <institution>Data and Web Science Group, University of Mannheim</institution>
          ,
          <country country="DE">Germany</country>
        </aff>
        <aff id="aff5">
          <label>5</label>
          <institution>Department of Computer and Information Science, Linko ̈ping University</institution>
          ,
          <addr-line>Linko ̈ping</addr-line>
          ,
          <country country="SE">Sweden</country>
        </aff>
        <aff id="aff6">
          <label>6</label>
          <institution>Inria Paris</institution>
          ,
          <country country="FR">France</country>
        </aff>
        <aff id="aff7">
          <label>7</label>
          <institution>Institut de Recherche en Informatique de Toulouse</institution>
          ,
          <country country="FR">France</country>
        </aff>
        <aff id="aff8">
          <label>8</label>
          <institution>Karlsruhe Institute of Technology</institution>
          ,
          <addr-line>Karlsruhe</addr-line>
          ,
          <country country="DE">Germany</country>
        </aff>
        <aff id="aff9">
          <label>9</label>
          <institution>LASIGE</institution>
          ,
          <addr-line>Faculdade de Cie</addr-line>
        </aff>
        <aff id="aff10">
          <label>10</label>
          <institution>National Institute of Technology Kurukshetra</institution>
          ,
          <addr-line>Haryana</addr-line>
          ,
          <country country="IN">India</country>
        </aff>
        <aff id="aff11">
          <label>11</label>
          <institution>Prague University of Economics and Business</institution>
          ,
          <country country="CZ">Czech Republic</country>
        </aff>
        <aff id="aff12">
          <label>12</label>
          <institution>Stibo Systems</institution>
          ,
          <country country="DK">Denmark</country>
        </aff>
        <aff id="aff13">
          <label>13</label>
          <institution>Swedish e-Science Research Centre</institution>
          ,
          <addr-line>Linko ̈ping</addr-line>
          ,
          <country country="SE">Sweden</country>
        </aff>
        <aff id="aff14">
          <label>14</label>
          <institution>Technical University of Denmark</institution>
          ,
          <country country="DK">Denmark</country>
        </aff>
        <aff id="aff15">
          <label>15</label>
          <institution>Trentino Digitale SpA</institution>
          ,
          <addr-line>Trento</addr-line>
          ,
          <country country="IT">Italy</country>
        </aff>
        <aff id="aff16">
          <label>16</label>
          <institution>Univ. Grenoble Alpes</institution>
          ,
          <addr-line>Inria, CNRS, Grenoble INP, LIG, F-38000 Grenoble</addr-line>
          ,
          <country country="FR">France</country>
        </aff>
        <aff id="aff17">
          <label>17</label>
          <institution>Universite ́ Co</institution>
        </aff>
        <aff id="aff18">
          <label>18</label>
          <institution>te d'Azur</institution>
          ,
          <addr-line>Inria, CNRS, I3S, Sophia Antipolis</addr-line>
          ,
          <country country="FR">France</country>
        </aff>
      </contrib-group>
      <pub-date>
        <year>2025</year>
      </pub-date>
      <volume>2</volume>
      <issue>2020</issue>
      <fpage>73</fpage>
      <lpage>126</lpage>
      <abstract>
        <p>The Ontology Alignment Evaluation Initiative (OAEI) aims at comparing ontology matching systems on precisely dened test cases. These test cases can be based on ontologies of dierent levels of complexity and use dierent evaluation modalities. The OAEI 2025 campaign oered 12 tracks and was attended by 20 participants. This paper is an overall presentation of that campaign.</p>
      </abstract>
    </article-meta>
  </front>
  <body>
    <sec id="sec-1">
      <title>1. Introduction</title>
      <p>
        The Ontology Alignment Evaluation Initiative1 (OAEI) is a coordinated international initiative, which
organizes the evaluation of ontology matching systems [
        <xref ref-type="bibr" rid="ref1 ref2">1, 2</xref>
        ], and has been run for 20 years now. The
main goal of the OAEI is to compare systems and algorithms openly and on the same basis to allow
anyone to conclude the best ontology matching strategies. Furthermore, the ambition is that from
such evaluations, developers can improve their systems and oer better tools addressing the evolving
application needs.
      </p>
      <p>
        The rst two events were organized in 2004: (i) the Information Interpretation and Integration
Conference (I3CON) held at the NIST Performance Metrics for Intelligent Systems (PerMIS) workshop
and (ii) the Ontology Alignment Contest held at the Evaluation of Ontology-based Tools (EON) workshop
of the annual International Semantic Web Conference (ISWC) [
        <xref ref-type="bibr" rid="ref3">3</xref>
        ]. Then, a unique OAEI campaign
occurred in 2005 at the workshop on Integrating Ontologies held in conjunction with the International
Conference on Knowledge Capture (K-Cap) [
        <xref ref-type="bibr" rid="ref4">4</xref>
        ]. From 2006 until the present, the OAEI campaigns were
held at the Ontology Matching workshop, co-located with ISWC [
        <xref ref-type="bibr" rid="ref5">5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16,
17, 18, 19, 20, 21, 22, 23</xref>
        ], which this year took place in Nara, Japan.2
      </p>
      <p>Since 2011, we have been using an environment for automatically processing evaluations that was
developed within the SEALS (Semantic Evaluation At Large Scale) project.3 SEALS provided a soware
infrastructure to automatically execute evaluations and evaluation campaigns for typical semantic
web tools, including ontology matching. During 2017 to 2023, a novel evaluation environment called
HOBBIT [24] was adopted for the HOBBIT Link Discovery track, and later extended to enable the
evaluation of other tracks. Some tracks are run exclusively through SEALS and others through HOBBIT,
but several allow participants to choose their preferred platform. Since 2022, the MELT framework [25]
has been adopted to facilitate the SEALS and HOBBIT wrapping and evaluation. Since 2023, most tracks
have adopted MELT as their evaluation platform.</p>
      <p>This paper synthesizes the 2025 evaluation campaign and introduces the results provided in the
participants’ papers. The remainder of the paper is organized as follows: in Section 2, we present the
overall evaluation methodology; in Section 3, we present the tracks and datasets; in Section 4 we present
and discuss the results; and nally, Section 5 discusses the lessons learned.</p>
    </sec>
    <sec id="sec-2">
      <title>2. Methodology</title>
      <sec id="sec-2-1">
        <title>2.1. Evaluation platforms</title>
        <p>The OAEI evaluation was conducted in one of two alternative platforms: the SEALS client, or the MELT
framework. Both of them have the goal of ensuring reproducibility and comparability of the results
across matching systems. As of this campaign, the use of the SEALS client and packaging format is
deprecated in favor of MELT, with the sole exception of the Interactive Matching track (see Section 3.5),
as simulated interactive matching is not yet supported by MELT.</p>
        <p>The SEALS client was developed in 2011. It is a Java-based command line interface for ontology
matching evaluation, which requires system developers to implement an interface and to wrap their
tools in a predened way, including all required libraries and resources.</p>
        <p>The MELT framework4 [25] was introduced in 2019 and is under active development. It allows
the development, evaluation, and packaging of matching systems for evaluation interfaces like SEALS
or HOBBIT. It further enables developers to use Python or any other programming language in their
matching systems, which, beforehand, had been a hurdle for OAEI participants. The evaluation client5
allows organizers to evaluate packaged systems whereby multiple submission formats are supported
(SEALS packages or matchers implemented as Web services). Starting from OAEI 2023, the MELT
framework also supports the SSSOM [26] format. Therefore, systems producing an alignment in the
SSSOM format can be evaluated as well.</p>
        <p>All platforms compute the standard evaluation metrics against the reference alignments: precision,
recall, and F-measure. In test cases requiring dierent evaluation modalities, the evaluation was carried
out a posteriori, using the alignments produced by the matching systems.</p>
      </sec>
      <sec id="sec-2-2">
        <title>2.2. Submission formats</title>
        <p>As already mentioned above, two submission formats were allowed: (1) SEALS package, and (2) MELT.
With the increasing usage of other programming languages than Java and increasing hardware
re2http://om.ontologymatching.org/2025
3http://www.seals-project.eu
4https://github.com/dwslab/melt
5https://dwslab.github.io/melt/matcher-evaluation/client
quirements for matching systems, since 2021, the MELT Web interface has been introduced to address
this issue. It mainly consists of a technology-independent HTTP interface6 which participants can
implement as they wish. Alternatively, they can use the MELT framework to assist them, as it can be
used to wrap any matching system as a docker container that implements the HTTP interface.</p>
        <p>Since 2024, we also allowed submission of alignment les in addition to the executable system in
case it requires substantial hardware or soware resources.</p>
      </sec>
      <sec id="sec-2-3">
        <title>2.3. OAEI campaign phases</title>
        <p>As in previous years, the OAEI 2025 campaign was divided into three phases: preparatory, execution,
and evaluation.</p>
        <p>In the preparation phase, the test cases were provided to participants during an initial evaluation
period between June 30ℎ and July 31, 2025. The goal of this phase is to ensure that the test cases
make sense to the participants and give them the opportunity to provide feedback to organizers on the
test case, as well as potentially report errors. At the end of this phase, the nal test base was frozen and
released.</p>
        <p>During the subsequent execution phase, participants test and potentially develop their matching
systems to automatically match the test cases. Participants can self-evaluate their results either by
comparing their output with the reference alignments or by using either of the evaluation platforms.
They can tune their systems with respect to the non-blind evaluation as long as they respect the rules
of the OAEI. Participants were required to register their systems by July 31 and make a preliminary
evaluation by August 31. The execution phase was terminated on September 30ℎ, 2025, at which date
participants had to submit the (near) nal versions of their systems.</p>
        <p>During the evaluation phase, systems were evaluated by all track organizers. In case minor problems
were found during the initial stages of this phase, they were reported to the developers, who were
given the opportunity to x and resubmit their systems. Initial results were provided directly to the
participants, whereas nal results for most tracks were published on the respective OAEI web pages
before the workshop.</p>
      </sec>
    </sec>
    <sec id="sec-3">
      <title>3. Tracks and Test Cases</title>
      <p>This year’s OAEI campaign consisted of 12 tracks, all of them including OWL ontologies, while only
one also including SKOS thesauri. They can be grouped into:
– Schema matching tracks, which have as objective matching ontology classes and/or properties.
– Instance matching tracks, which have as objective matching ontology instances.
– Instance and schema matching tracks, which involve both of the above.
– Complex matching tracks, which have as objective nding complex correspondences between
ontology entities.
– Interactive tracks, which simulate user interaction to enable the benchmarking of interactive
matching algorithms.</p>
      <sec id="sec-3-1">
        <title>The tracks are summarized in Table 1 and detailed in the following sections.</title>
      </sec>
      <sec id="sec-3-2">
        <title>6https://dwslab.github.io/melt/matcher-packaging/web</title>
        <p>3.1. Anatomy</p>
        <p>=
=, &lt;=
=
=
=, &lt;=
=
=
=
relations
confidence modalities
language</p>
        <p>SEALS MELT
The anatomy track comprises a single test case consisting of matching two fragments of biomedical
ontologies which describe the human anatomy7 (3304 classes) and the anatomy of the mouse8 (2744
classes). The evaluation is based on a manually curated reference alignment. This dataset has been used
since 2007 with some improvements over the years [27].</p>
        <p>Systems are evaluated with the standard parameters of precision, recall, F-measure. Additionally,
recall+ is computed by excluding trivial correspondences (i.e., correspondences that have the same
normalized label). Alignments are also checked for coherence using the Pellet reasoner. The evaluation
was carried out on a machine with a 5 core CPU @ 1.80 GHz with 16GB allocated RAM, using the
MELT framework. For some systems, the SEALS client has been used. However, the evaluation
parameters were computed a posteriori, aer removing from the alignments produced by the systems,
correspondences expressing relations other than equivalence, as well as trivial correspondences in the
oboInOwl namespace (e.g., oboInOwl#Synonym = oboInOwl#Synonym). The results obtained with the
SEALS client vary in some cases by 0.5% compared to the results presented in Section 4.2.</p>
        <sec id="sec-3-2-1">
          <title>3.2. Conference</title>
          <p>The conference track consists of a suite of 21 matching tasks corresponding to the pairwise combination
of 7 moderately expressive ontologies describing the domain of organizing conferences. The dataset
and its usage is described in [28].</p>
          <p>The track uses several reference alignments for evaluation: the old (and not fully complete) manually
curated open reference alignment, ra1; an extended, also manually curated version of this alignment,
ra2; a version of the latter corrected to resolve violations of conservativity, rar2; and an uncertain
version of ra1 produced through crowd-sourcing, where the score of each correspondence is the fraction
of people in the evaluation group that agree with the correspondence. The latter reference was used in
two evaluation modalities: discrete and continuous evaluation. In the former, correspondences in the
uncertain reference alignment with a score of at least 0.5 are treated as correct whereas those with
lower score are treated as incorrect, and standard evaluation parameters are used to evaluated systems.</p>
        </sec>
      </sec>
      <sec id="sec-3-3">
        <title>7https://www.cancer.gov/cancertopics/cancerlibrary/terminologyresources 8http://www.informatics.jax.org/searches/AMA form.shtml</title>
        <p>In the latter, weighted precision, recall and F-measure values are computed by taking into consideration
the actual scores of the uncertain reference, as well as the scores generated by the matching system. For
the sharp reference alignments (ra1, ra2 and rar2), the evaluation is based on the standard parameters,
as well the F0.5-measure and F2-measure and on conservativity and consistency violations. Whereas F1
is the harmonic mean of precision and recall where both receive equal weight, F2 gives higher weight
to recall than precision and F0.5 gives higher weight to precision higher than recall. The second test
case contains open reference alignment and systems were evaluated using the standard metrics.</p>
        <p>Two baseline matchers are used to benchmark the systems: edna string edit distance matcher; and
StringEquiv string equivalence matcher as in the anatomy test case.</p>
        <sec id="sec-3-3-1">
          <title>3.3. Multifarm</title>
          <p>The multifarm track [29] aims at evaluating the ability of matching systems to deal with ontologies in
dierent natural languages. This dataset results from the translation of 7 ontologies from the conference
track (cmt, conference, confOf, iasted, sigkdd, ekaw and edas) into 12 languages: Arabic (ar), Chinese
(cn), Czech (cz), Dutch (nl), French (fr), German (de), Italian (it), Portuguese (pt), Russian (ru), Spanish
(es), Hindi(hi), and Turkish (tr). The dataset is composed of 55 pairs of languages, with 49 matching tasks
for each of them, taking into account the alignment direction (e.g., cmt →edas and cmt →edas
are distinct matching tasks). While part of the dataset is openly available, all matching tasks involving
the edas and ekaw ontologies (resulting in 55 × 24 matching tasks) are used for blind evaluation.</p>
          <p>We consider two test cases: i) those tasks where two dierent ontologies (cmt→edas, for instance) have
been translated into two dierent languages; and ii) those tasks where the same ontology (cmt→cmt)
has been translated into two dierent languages. For the tasks of type ii), good results are not only
related to the use of specic techniques for dealing with cross-lingual ontologies, but also on the
ability to exploit the identical structure of the ontologies. This year, we report the results on dierent
ontologies (i).</p>
          <p>The reference alignments used in this track derive directly from the manually curated Conference
ra1 reference alignments. In 2021, alignments have been manually evaluated by domain experts. The
evaluation is blind. The systems have been executed on a Windows machine congured with 16GB of
RAM running under a Intel Core CPU 2.00GHz x8 cores. The evaluation was performed using the MELT
platform. Every participating system was executed in its standard setting and we compare precision,
recall and F-measure as well as the computation time.</p>
        </sec>
        <sec id="sec-3-3-2">
          <title>3.4. Complex Matching</title>
          <p>The complex matching track is meant to evaluate the matchers based on their ability to generate complex
alignments. A complex alignment is composed of complex correspondences typically involving more
than two ontology entities, such as 1:AcceptedPaper ≡  2:Paper ⊓ ∃2:hasDecision.2:Acceptance.</p>
          <p>The track ran with eight datasets: Conference and Populated Conference, Hydrography, GeoLink
and Populated GeoLink, Populated Enslaved, and Biomedical, a new dataset introduced this year for
complex multi-ontology matching.</p>
          <p>The Conference dataset comprises three ontologies: cmt, conference, and ekaw from the conference
dataset. The reference alignment was created as a consensus between experts. To allow matchers which
rely on instances to participate over the Conference complex track, the Populated Conference dataset
is composed of 5 conference ontologies populated with more or less common instances, resulting in
6 datasets: (6 versions on the repository: v0, v20, v40, v60, v80 and v100). Details on the population
and evaluation modalities are available.9 The Hydrography dataset is composed of four tasks, where
four source ontologies (Hydro3, HydrOntology native, HydrOntology translated, and Cree) are aligned
with a single target ontology, the Surface Water Ontology (SWO). The GeoLink dataset includes a
single matching task between the GeoLink Base Ontology (GBO) and the GeoLink Modular Ontology
(GMO) [30]. The Populated GeoLink dataset is based on the previous one and adds instance data from
9https://framagit.org/IRIT UT2J/conference-dataset-population
seven data repositories in the GeoLink project [31]. The Populated Enslaved dataset is composed
of two ontologies, the Enslaved Ontology and the Enslaved Wikidata Knowledge Graph, where the
instance data originates from the Wikidata repository and the consensus was obtained from domain
experts from several historian research institutions [32]. The Biomedical dataset is a new addition
this year for the specic task of complex multi-ontology matching, and it is based on the existing
logical denitions from three dierent biomedical ontologies: the Human Phenotype ontology (HP), the
Mammalian Phenotype ontology (MP), and the Worm Phenotype ontology (WBP) [33]. Both HP and
MP have the same set of target ontologies: Cell ontology (CL), Chemical Entities of Biological Interest
(ChEBI), Gene Ontology (GO), Phenotype and Trait Ontology (PATO), and Uber Anatomy Ontology
(UBERON). WBP uses ChEBI, GO, PATO, and the C.elegans Gross Anatomy Ontology (WBbt).</p>
          <p>The participants of the track output their (complex) correspondences in the EDOAL format. For the
Conference dataset, the complex correspondences are manually compared to the ones of the consensus
alignment. Three new evaluation strategies were debuted this year: Class evaluation [34], Graph Edit
Distance [35], and Tree Edit Distance [36], which were run in the remaining tasks.</p>
        </sec>
        <sec id="sec-3-3-3">
          <title>3.5. Interactive Matching</title>
          <p>The interactive matching track aims to assess the performance of semi-automated matching systems by
simulating user interaction [37, 38, 39]. The evaluation thus focuses on how interaction with the user
improves the matching results. Currently, this track does not evaluate the user experience or the user
interfaces of the systems [40, 38].</p>
          <p>The interactive matching track is based on the datasets from the Anatomy and Conference tracks,
which have been previously described. It relies on the SEALS client’s Oracle class to simulate user
interactions. An interactive matching system can present a collection of correspondences simultaneously
to the oracle, telling the system whether that correspondence is correct or not. If a system presents
up to three correspondences together and each correspondence presented has a mapped entity (i.e.,
class or property) in common with at least one other correspondence presented, the oracle counts this
as a single interaction, under the rationale that this corresponds to a scenario where a user is asked
to choose between conicting candidate correspondences. To simulate the possibility of user errors,
the oracle can be set to reply with a given error probability (randomly, from a uniform distribution).
We evaluated systems with four dierent error rates: 0.0 (perfect user), 0.1, 0.2, and 0.3. In addition to
the standard evaluation parameters, we also compute the number of requests made by the system, the
total number of distinct correspondences asked, the number of positive and negative answers from the
oracle, the performance of the system according to the oracle (to assess the impact of the oracle errors
on the system) and nally, the performance of the oracle itself (to assess how erroneous it was).</p>
          <p>The evaluation was carried out on a server with 3.46 GHz (6 cores) and 8GB RAM allocated to the
matching systems. For systems requiring more RAM, the evaluation was carried out on a computer
with an AMD Ryzen 7 5700G 3.80 GHz CPU and 32GB RAM, with 10GB of max heap space allocated
to Java. Each system was run ten times and the nal result of a system for each error rate represents
the average of these runs. For the Conference dataset with the ra1 alignment, precision and recall
correspond to the micro-average over all ontology pairs, whereas the number of interactions is the total
number of interactions for all the pairs.
3.6. Bio-ML
The Bio-ML track [41] incorporates equivalence ontology matching (OM) tasks for biomedical ontologies,
with ground truth (equivalence) mappings extracted from Mondo [42] and UMLS [43] (see Table 2).
Mondo aims to integrate disease concepts worldwide, while UMLS is a meta-thesaurus for the biomedical
domain. Based on techniques (ontology pruning, subsumption mapping construction, negative candidate
mapping generation, etc.) proposed in [41], we make available ve OM pairs with their information
reported in Table 3. Each OM pair is accompanied with equivalence matching tasks; each matching task
has two data split settings, i.e., unsupervised setting with no training mappings, and semi-supervised
Ontology Pair
OMIM-ORDO</p>
          <p>NCIT-DOID
SNOMED-FMA
SNOMED-NCIT
SNOMED-NCIT</p>
          <p>Category
Disease
Disease
Body
Pharm
Neoplas
setting with 30% ground truth mappings for training/validation. Since the 2023 edition, Bio-ML has
added a logical module enrichment [44] to add entities to the pruned ontologies to provide more context
for alignment, annotated as “not used in alignment” and ignored in evaluation. For evaluation, in [41] we
proposed both global matching and local ranking; the former aims to evaluate the overall performance
by computing Precision, Recall, and F1 metrics for the output mappings against the reference mappings,
while the latter aims to evaluate the ability to distinguish the correct mapping out of several challenging
negatives by ranking metrics Hits@K and MRR.</p>
          <p>We adopted a exible way of evaluating participating systems. First, participants can freely choose
any tasks and settings they would like to attend. Second, for systems that have been well-adapted to
the MELT platform, we used MELT to produce the output mappings. Third, for systems that have been
implemented elsewhere and are not easy to be made compatible with MELT, we used their source code.
Fourth, we also allowed participants (with trust) to directly upload output mappings if their systems had
not been published and had not been made compatible with MELT. In the nal result tables, we used
superscripts †, ‡, and * to indicate that the results came from MELT, source code implementation, and
direct result submission, respectively. All our evaluations were conducted with the DeepOnto12 [45]
library.</p>
        </sec>
        <sec id="sec-3-3-4">
          <title>3.7. Digital Humanities</title>
          <p>
            The use of controlled vocabularies is widespread within the digital humanities (DH) [46]. The
development and usage of these vocabularies by dierent parties in related domains naturally leads to overlaps
in content [
            <xref ref-type="bibr" rid="ref2">2</xref>
            ]. While ontology matching helps with alignment and integration tasks, the application of
these systems to the digital humanities poses special challenges. Highly specic domain terminology
oen leads to smaller vocabularies, which oentimes include multiple (ancient) languages. Furthermore,
matching systems need to be compatible with SKOS vocabularies, since their use is fairly common
within the community.
          </p>
          <p>The DH track participated for the second time. It includes eight test cases from archaeology, cultural
history and DH / computer science. Each test case consists of two SKOS (using RDF/XML as syntax)
vocabularies to be matched and one manually created gold standard reference. For details on the nine
source vocabularies and on the test cases, see Table 4 and Table 5.
10Created from OMIM texts by Mondo’s pipeline tool avaiable at: https://github.com/monarch-initiative/omim.
11Created by the ocial snomed-owl-toolkit available at: https://github.com/IHTSDO/snomed-owl-toolkit.
12https://krr-oxford.github.io/DeepOnto/#/</p>
          <p>The evaluation was executed on a virtual machine with 8 cores (2.4GHz each) and 16 GB RAM. To
quantify the performance, precision, recall and F1-score were used, while only evaluating equivalence
relationships. If matching systems resulted in either errors or zero identied matches, the task was
considered as failed. Adhering to the OAEI rules, no settings were changed before running the matching
systems.</p>
        </sec>
        <sec id="sec-3-3-5">
          <title>3.8. Archaeology Multilingual</title>
          <p>The archaeology multilingual track is based on an archaeology test case of the digital humanities track,
see Section 3.7, with focus on evaluating matcher performance when dealing with dierent languages.</p>
          <p>Like the DH track, this track participated for the second time. Each test case uses iDAI.world and
PACTOLS (see Table 4 for more information) as source resp. target. Both vocabularies contain terms in
English, French, German, and Italian. To create the ten test cases, all but one language were removed
from both vocabularies, leading to 10 dierent test cases, consisting of two monolingual vocabularies
and a manually created gold standard reference.</p>
          <p>The evaluation modalities are identical to ones in the digital humanities track, see Section 3.7.
13This is the eld to which the CV was grouped within our dataset.
14This is the number of concepts in the primary language of the CV before any preprocessing steps.
15https://vocabs.dariah.eu/defc thesaurus/en/
16https://isl.ics.forth.gr/bbt-federated-thesaurus/PACTOLS/en/
17https://vocabs.dariah.eu/iad thesaurus/en/
18https://isl.ics.forth.gr/bbt-federated-thesaurus/DAI/en/
19https://vocabs.dariah.eu/parthenos vocabularies/en/
20https://vocabs.acdh.oeaw.ac.at/oeai-cp/en/
21https://vocabs.dariah.eu/dha taxonomy/en/
22https://vocabularies.unesco.org/browser/thesaurus/en/
23https://vocabs.dariah.eu/tadirah/en/
24The number of terms varies depending on the branch used for the respective domain.</p>
        </sec>
        <sec id="sec-3-3-6">
          <title>3.9. Circular Economy</title>
          <p>In recent years, the Circular Economy (CE) domain has shown interest in representing domain knowledge
using ontologies. Since there are some existing CE-specic ontologies with more emerging, providing
alignments among ontologies can enhance the interoperability and reusability of such ontologies. The
circular economy track was proposed since 2024, and included 2 tasks in this year. In both tasks
CE-specic ontology is matched to the Circular Economy Ontology Network (CEON) [47]. The rst task
is to match CEON to the Sustainable Bioeconomy and Bioproducts Ontology (BiOnto) [48]. The second
task is to match CEON to materials domain ontology, MatOnto [49]. CEON (including 214 classes) from
the Onto-DESIDE project,25 aims to represent core concepts for the CE domain [47]. BiOnto (including
780 classes) from the BIOVOICES project,26 focuses on establishing a shared and common terminology
in the bioeconomy domain. MatOnto (848 classes) covers the materials domain. Materials are a central
concept in circular value networks. In all, using BiOnto or MatOnto dierent stakeholders participating
circular value networks can provide information according to ontologies [50].</p>
          <p>The evaluation was conducted over standard parameters which are precision, recall, f-measure and
alignment size. The reference alignment for the matching task was initially done in [51, 52] and further
validated by ontology engineers and CE domain experts from Onto-DESIDE project. The results is
presented in Section 4.10.</p>
        </sec>
        <sec id="sec-3-3-7">
          <title>3.10. Beyond Equivalence</title>
          <p>This is the rst time the Beyond Equivalence track is being organized. The goal of this track is to
evaluate the ability of ontology matching systems to detect correspondences beyond simple equivalence.
Specically, systems are tasked with identifying ve mutually disjoint relation types: Equivalence (≡ ),
Superclass of (≤), Subclass of (≥), Overlap (≃), and Disjointness (⊥).</p>
          <p>Benchmark data for the track is drawn from two main sources. First, a variety of industrial product
classication schemes (e.g., GPC,27 UNSPSC,28 ETIM,29, and eClass30), collectively referred to as “Product
Classication Standards”. Second, a diverse set of test-cases derived from the STROMA/TaSeR [53, 54]
repository, consists of d 5 datasets: g1-web, g2–diseases, g3–text, g5–groceries, and g7–literature.
Altogether, the benchmark suite comprises multiple datasets of varying size, structure, and semantic
complexity; in each case the reference alignments are annotated with explicit relation types beyond
simple equivalence. Table 6 presents the statistic information of the datasets.</p>
          <p>Beyond Equivalence track encourages the development of more powerful and semantically-aware
matching systems — in line with applications such as knowledge-graph merging, master data
integration, and semantic search, where ontologies frequently dier in granularity, structure, or conceptual
perspective.</p>
        </sec>
        <sec id="sec-3-3-8">
          <title>3.11. Knowledge Graph</title>
          <p>The Knowledge Graph track was run for the h year. The task of the track is to match pairs of
knowledge graphs whose schema and instances have to be matched simultaneously. The individual
knowledge graphs are created by running the DBpedia extraction framework on eight dierent Wikis
from the Fandom Wiki hosting platform31 in the course of the DBkWik project [55, 56]. They cover
dierent topics (movies, games, comics, and books) and three Knowledge Graph clusters sharing the
same domain e.g., star trek, as shown in Table 7.</p>
          <p>The evaluation is based on reference correspondences at both schema and instance levels. While the
schema-level correspondences were created by experts, the instance correspondences were extracted
25https://ontodeside.eu
26https://www.biovoices.eu
27https://gpc-browser.gs1.org/
28https://www.undp.org/unspsc
29https://www.etim-international.com/
30https://eclass.eu/en/
31https://www.wikia.com/
from the wiki page itself. Due to the fact that not all interwiki links on a page represent the same
concept, a few restrictions were made: 1) only links in sections with a header containing “link” are used,
2) all links are removed where the source page links to more than one concept in another wiki (ensures
the alignments are functional), 3) multiple links which point to the same concept are also removed
(ensures injectivity), 4) links to disambiguation pages were manually checked and corrected. Since we
do not have a correspondence for each instance, class, and property in the graphs, this gold standard is
only a partial gold standard.</p>
          <p>The evaluation was executed on a virtual machine (VM) with 32GB of RAM and 16 vCPUs (2.4 GHz),
with Debian 9 operating system and Openjdk version 1.8.0 265. For evaluating all possible submission
formats, MELT framework is used. The corresponding code for evaluation can be found on Github.32</p>
          <p>The alignments were evaluated based on precision, recall, and F-measure for classes, properties, and
instances (each in isolation). The partial gold standard contained 1:1 correspondences, and we further
assume that in each knowledge graph, only one representation of the concept exists. This means that if
we have a correspondence in our gold standard, we count a correspondence to a dierent concept as a
false positive. The count of false negatives is only increased if we have a 1:1 correspondence and it is
not found by a matcher.</p>
          <p>As a baseline, we employed two simple string-matching approaches. The source code for these
matchers is publicly available.33</p>
        </sec>
        <sec id="sec-3-3-9">
          <title>3.12. Pharmacogenomics</title>
          <p>In 2025, the Pharmacogenomics track was run for the third time. This track focuses on matching
knowledge units from the pharmacogenomics domain. These units are -ary tuples – so-called
“pharmacogenomic relationships” – and involve drugs, genetic factors, and phenotypes (see Figure 1). A
pharmacogenomic tuple states that patients being treated by the specied drugs while having the
specied genetic factors may experience the given phenotypes.
32https://github.com/dwslab/melt/tree/master/examples/kgEvalCli
33http://oaei.ontologymatching.org/2019/results/knowledgegraph/kgBaselineMatchers.zip
{d1, . . . , d}
{gf1, . . . , gf}</p>
          <p>In the Semantic Web formalisms, only binary predicates exist. That is why pharmacogenomic tuples
are reied: tuples become individuals that are linked to their components with binary predicates
(Figure 1(c)). Hence, the task of matching pharmacogenomic tuples is [57]:
– An instance matching task that aims at nding alignments between individuals representing
reied tuples;
– A structure-based matching task in which neighbors of reied tuples are compared to conclude on
the potential alignment between tuples. Recall that the only available information about these
tuples is their neighbors (e.g., no labels, or other properties).</p>
          <p>To illustrate, two tuples associating the same sets of drugs, genetic factors, and phenotypes have the
same neighbors, thus represent the same two “pharmacogenomics relationships”, and thus should be
detected as identical.</p>
          <p>Beside the arity of tuples, matchers need to face issues such as incompleteness (e.g., missing drugs)
and heterogeneity (e.g., a gene version like CYP2C9*4 is more specic than the gene itself CYP2C9, the
phenotype hemorrhagee is more specic than the phenotype vascular disorders). Dierent types
of alignments are thus expected to be identied between pharmacogenomic tuples, which is somehow
unusual in an instance matching task. The Pharmacogenomics track features the identication of
identical tuples (=), equivalent tuples (Close), tuples being more specic (&lt;) or more general (&gt;) than
others, and tuples being related to some extent (Related). See [57, 58] for a detailed denition of these
dierent alignment types between individuals.</p>
          <p>To perform this alignment task, matchers can rely on additional background knowledge about
components of pharmacogenomic tuples. This knowledge includes ontology classes instanciated by
the components of tuples (i.e., drugs, genetic factors, phenotypes) and their hierarchical organization,
partOf links between gene versions and genes, sameAs links between identical drugs, genes, or
phenotypes, and dependsOn links between complex phenotypes and their components (e.g.,
“warfarininduced bleeding” depends on “warfarin” and on “bleeding”).</p>
          <p>To evaluate matchers and their scalability, the Pharmacogenomics track comprises three tasks
involving respectively 10, 50, and 100% of the 50,435 pharmacogenomic tuples represented within the
PGxLOD knowledge graph34 [59]. For each task, the selected pharmacogenomic tuples are evenly split
into two ontologies to match. To take into account the specicity of the dierent alignment types that
are expected, matchers are evaluated through two settings:
Fine-grained setting Only alignments of the exact type expected in the reference are considered
correct. To illustrate, an output alignment (1, =, 2) where (1, Close, 2) was expected will be
considered as incorrect. Precision, Recall, and F1-score are computed for each type of alignment.
Coarse-grained setting Any type of alignment between entities expected to be aligned will be
considered as correct. To illustrate, an output alignment (1, =, 2) where (1, Close, 2) was expected
will be considered as correct. Precision, Recall, and F1-score are computed globally accordingly.
4. Results and Discussion
4.1. Participation
al
u
ig
tin
l
ul
SMatch SMatch-M
L L
#
#
#
#
#
#
#
#
#
#
#
1
#
#
#
#
#
#
#
#
#
3</p>
          <p>r
tcha Mape
a D
M M
#
#
#
#
#
#
#
#
#
#
#G
#
Following an initial period of growth, the number of OAEI participants has remained approximately
constant since 2012. This year we count with 20 participating systems. Table 8 lists the participants
and the tracks in which they competed. It is worth mentioning that the Bio-ML track has additional
participants (e.g., BERTMap [60] and BERTSubs [61]) that are not counted in the number of participants.
This is because they need training and validation which are not yet fully supported by the OAEI
evaluation platforms, and thus they were tested locally with Bio-ML results reported, but without an
OAEI system submission. Some matching systems participated with dierent variants (e.g., LogMap
and LSMatch), whereas others were evaluated with dierent congurations, as requested by developers
(see test case sections for details). The following sections summarize the results for each track.</p>
        </sec>
        <sec id="sec-3-3-10">
          <title>4.2. Anatomy</title>
          <p>The results for the Anatomy track are shown in Table 9. Among the 11 systems participating in
the Anatomy track, 10 achieved an F-measure higher than the StringEquiv baseline. Three systems
were rst-time participants (i.e., Agent-OM, DRAL-OA, LogMapLLM) in anatomy track. Long-term
participating systems showed few changes in comparison with previous years with respect to alignment
quality (precision, recall, F-measure, and recall+) and size. The exception were ALIN which increased
in F-measure (from 0.851 to 0.912), recall (from 0.75 to 0.884), recall+ (from 0.489 to 0.7), size (from 1156
to 1423), and decreased in precision (from 0.984 to 0.942). LogMap-Bio increased in size (from 1549 to
1561), recall (from 0.908 to 0.911), recall+ (from 0.757 to 0.766), and decreased in precision (from 0.888
to 0.885). MDMapper increased in size (from 1441 to 1483), recall+ (from 0.703 to 0.707), and decreased
in precision (from 0.926 to 0.899), F-measure (from 0.903 to 0.889), recall (from 0.881 to 0.879). In terms
of runtime, 5 out of 11 systems computed an alignment in less than 100 seconds. LogMapLt remains the
system with the shortest runtime. Regarding quality, Matcha achieved the highest F-measure (0.941)
and recall+ (0.82), but three other systems obtained an F-measure above 0.88 (Agent-OM, ALIN, and
LogMapLLM) which is at least as good as the best systems in OAEI 2007-2010. Like in previous years,
there is no signicant correlation between the quality of the generated alignment and the run time.
Four systems produced coherent alignments (i.e., LogMap, LogMap-Bio, LogMapKG and LogMapLLM).</p>
        </sec>
        <sec id="sec-3-3-11">
          <title>4.3. Conference</title>
          <p>The conference evaluation results using the sharp reference alignment rar2 are shown in Table 10.
For the sake of brevity, only results with this reference alignment and considering both classes and
properties are shown. For more detailed evaluation results, please check the conference track’s web
page.</p>
          <p>With regard to two baselines we can group tools according to system’s position: there are ve
matchers above (or equal to) edna baseline (ALIN, LogMap, Matcha, Agent-OM, and MDMapper), and
two matchers below edna baseline but above StringEquiv baseline (LogMapLt, and LSMatch). Two
matchers (MDMapper, and LSMatch) do not match properties at all.</p>
          <p>The performance of all matching systems regarding their precision, recall and F1-measure is plotted
in Figure 2. Systems are represented as squares or triangles, whereas the baselines are represented as
circles.</p>
          <p>The Conference evaluation results using the uncertain reference alignments are presented in Table 11.
Out of the 7 systems, ve (Agent-OM, ALIN, LogMapLt, LSMatch, and MDMapper) use 1.0 to all
correspondences. The remaining 2 systems (LogMap, Matcha) have a wide variation of condence
values. Agent-OM internally applied a threshold of 0.9 before submission of alignments.</p>
          <p>Prec. F0.5-m. F1-m. F2-m. Rec. Inc.Align. Conser.V. Consist.V.</p>
          <p>Systems using xed condences show clear gains from the sharp to the uncertain evaluations. As in
2024, the discrete metric particularly benets these systems by downweighting low-consensus matches.
ALIN, LogMapLt, and MDMapper maintain balanced precision and recall, while LSMatch achieves the
highest precision overall (0.88) and converts it into strong F-measure improvements. Agent-OM also
performs competitively, showing robust recall despite lower precision.</p>
          <p>Systems assigning graded condences continue to perform best under the continuous metric. Their
ability to model uncertainty enables stable precision and greater recall than in the sharp setting,
conrming the advantage of calibrated condence outputs.</p>
          <p>Overall, the 2025 results reinforce two trends observed in 2024: (1) recall gains under uncertain
evaluation are broader and more consistent, narrowing performance gaps among systems, and (2) both
xed- and graded-condence approaches can benet from uncertainty, provided alignments reect the
majority consensus.</p>
        </sec>
        <sec id="sec-3-3-12">
          <title>4.4. Multifarm</title>
          <p>This year, 4 systems have registered to participate in the Multifarm track: LogMap, LogMapLt, Matcha,
and LSMatch-Multilingual. The number of participating tools is similar with respect to the last 4
campaigns (4 in 2024, 4 in 2023, 5 in 2022, 6 in 2021). This year, we welcome back LSMatch-Multilingual.
The reader can refer to the OAEI papers for a detailed description of the strategies adopted by each
system.</p>
          <p>The Multifarm evaluation results based on the blind dataset are presented in Table 12, demonstrating
the aggregated results for the matching tasks. They have been computed using the MELT framework
without applying any threshold to the results. They are measured in terms of macro precision and recall.
The results of non-specic systems are not reported here, as we could observe in the last campaigns
that they can have intermediate results in tests of type ii) (same ontologies task) and poor performance
in tests i) (dierent ontologies task).</p>
          <p>The systems have been executed on a Windows machine congured with 16GB of RAM running
under a Intel Core i7-9750H @2.60Ghz CPU. All measurements are based on a single run. As for each
campaign, we observed large dierences in the time required for a system to complete the 55 x 24
matching tasks:</p>
          <p>The results (Table 12) indicate notable dierences in performance across the four systems (LogMap,
LogMapLt, Matcha, and LSMatch-Multilingual) with regard to processing time, precision, F-measure,
and recall. LogMap exhibits the shortest processing time (6.7 minutes) and achieves the highest precision
(0.87), but its recall is relatively low (0.10), resulting in a moderate F-measure of (0.18). LogMapLt
takes longer (16.9 minutes) but shows lower precision (0.84) and a minimal F-measure (0.008), along
with a low recall (0.004). Matcha requires even more time (408 minutes) and has a relatively balanced
performance, with a precision of 0.26, an F-measure of 0.26, and the highest recall among the systems
(0.25). Finally, LSMatch-Multilingual has the runtime of (36 minutes) with precision (0.79), best recall
(0.30), and hence best F-measure of 0.44, indicating limited eectiveness despite the extended processing
time. Overall, LogMap stands out for its eciency and higher precision, while LSMatch-Multilingual
demonstrates better recall, and F-measure.</p>
        </sec>
        <sec id="sec-3-3-13">
          <title>4.5. Complex Matching</title>
          <p>Regarding Conference dataset (non-populated) sub-track, we had only one participant: Matcha. Matcha
delivered simple equivalences, complex correspondences, and subsumptions. Within this track only
complex correspondences have been evaluated. With regard to complex correspondences no TPs
were identied. All complex correspondences contained intersections (of classes and also of classes
and properties). While the intersection of classes is an important construct, EDOAL does not allow
for directly intersecting a class and a property. Instead, class restrictions should be applied, such as
AttributeDomainRestriction.</p>
          <p>For the remaining tasks there were two participating systems: CMatch and Matcha. Additionally, the
alignments from previous participating systems were added as placeholder baselines: AMLC, AROA,
and CANARD (from the 2020 OAEI edition). The results for the evaluation metrics of Graph Edit
Distance (GED) and Tree Edit Distance (TED) are shown in Table 13. As not all systems participated in
all datasets it is dicult to draw general conclusions. CMatch has increased precision but low recall,
leading to an F-measure that is lower than other systems in most cases. CANARD appears to be the
more consistent choice across both metrics, with Matcha achieving far lower results with the TED
metric but comparable ones with the GED metric. AMLC achieves the worse results out of all systems,
and it is interesting to note that AROA outperforms all other systems in the single task it participated
in. As the submission was done as alignments, conclusions cannot be drawn on the computational and
runtime costs required by the systems.</p>
        </sec>
        <sec id="sec-3-3-14">
          <title>4.6. Interactive matching</title>
          <p>This year, two systems (ALIN and LogMap) participated in the Interactive matching track. Their results
are shown in Table 14 and Figure 3 for both the Anatomy and Conference datasets.</p>
          <p>The table includes the following information (column names within parentheses):
NI stands for non-interactive, and refers to the results obtained by the matching system in the original track.
– The performance of the system: Precision (Prec.), Recall (Rec.), and F-measure (F-m.) with respect
to the xed reference alignment, as well as Recall+ (Rec.+) for the Anatomy task. To facilitate the
assessment of the impact of user interactions, we also provide the performance results from the
original tracks, without interaction (line with Error NI).
– To ascertain the impact of the oracle errors, we provide the performance of the system with
respect to the oracle (i.e., the reference alignment as modied by the errors introduced by the
oracle: Precision oracle (Prec. oracle), Recall oracle (Rec. oracle) and F-measure oracle (F-m.
oracle). For a perfect oracle, these values match the actual performance of the system.
– Total requests (Tot Reqs.) represents the number of distinct user interactions with the tool, where
each interaction can contain one to three conicting correspondences, that could be analyzed
simultaneously by a user.
– Distinct correspondences (Dist. Mapps) counts the total number of correspondences for which
the oracle gave feedback to the user (regardless of whether they were submitted simultaneously,
or separately).
– Finally, the performance of the oracle itself with respect to the errors it introduced can be gauged
through the positive precision (Pos. Prec.) and negative precision (Neg. Prec.), which measure
respectively the fraction of positive and negative answers given by the oracle that are correct.</p>
          <p>For a perfect oracle, these values are equal to 1 (or 0, if no questions were asked).</p>
          <p>Figure 3 shows the time intervals between the questions to the user/oracle for the dierent systems
and error rates. Dierent runs are depicted with dierent colors.</p>
          <p>The matching systems that participated in this track employ dierent user-interaction strategies.
While LogMap uses user interactions exclusively in the post-matching steps to lter their candidate
correspondences, ALIN can also add new candidate correspondences to its initial set. LogMap requests
feedback on only selected correspondences candidates (based on their similarity patterns or their
involvement in unsatisabilities). ALIN and LogMap can both ask the oracle to analyze several conicting
correspondences simultaneously.</p>
          <p>The performance of the systems usually improves when interacting with a perfect oracle in
comparison with no interaction. The improvement of ALIN is mainly because of its high number of oracle
requests, and its non-interactive performance was the lowest of the interactive systems, and thus the
easiest to improve.</p>
          <p>Although system performance deteriorates when the error rate increases, there are still benets from
the user interaction—some of the systems’ measures stay above their non-interactive values even for
the larger error rates. Naturally, the more a system relies on the oracle, the more its performance tends
to be aected by the oracle’s errors.</p>
          <p>The impact of the oracle’s errors is linear for ALIN in most tasks, as the F-measure according to
the oracle remains approximately constant across all error rates. It is supra-linear for LogMap in all
datasets.</p>
          <p>Another aspect that was assessed, was the response time of systems, i.e., the time between requests.
Two models for system response times are frequently used in the literature [62]: Shneiderman and
Seow take dierent approaches to categorize the response times taking a task-centered view and a
user-centered view respectively. According to task complexity, Shneiderman denes response time in
four categories: typing, mouse movement (50-150 ms), simple frequent tasks (1 s), common tasks (2-4 s)
and complex tasks (8-12 s). While Seow’s denition of response time is based on the user expectations
towards the execution of a task: instantaneous (100-200 ms), immediate (0.5-1 s), continuous (2-5 s),
captive (7-10 s). Ontology alignment is a cognitively demanding task and can fall into the third or fourth
categories in both models. In this regard the response times (request intervals as we call them above)
observed in all datasets fall into the tolerable and acceptable response times, and even into the rst
categories, in both models. The request intervals for LogMap and ALIN stay at a few milliseconds for
most datasets. It could be the case, however, that a user would not be able to take advantage of these
low response times because the task complexity may result in higher user response time (i.e., the time
the user needs to respond to the system aer the system is ready).
4.7. Bio-ML
Our results comprise ve tables, where each table corresponds to a specic ontology matching (OM)
pair and includes results for both the unsupervised and semi-supervised settings. An overview of the
results is presented in Table 15. Full results are available on the OAEI 2025 Bio-ML website.35</p>
          <p>Briey, the participating systems are: (i) machine learning-based approaches, including
Agent-OM [63], BERTMap, BERTMapLt [60], BioGITOM [64], BioSTransMatch, LogMap-LLM,
Logmap+OWL2Vec4OA, Matcha [65, 66], and OWL2Vec4OA; and (ii) traditional symbolic systems,
namely LogMap, LogMap-Bio, and LogMapLt [44].</p>
          <p>The top-performing systems varied across tasks. BioGITOM achieved the highest F1 score in 3 out of
5 semi-supervised tasks, with BERTMap and LogMap-LLM each leading in one. For unsupervised tasks,
LogMap-Bio and LogMap-LLM attained the best F1 score in 2 tasks each, with BERTMap leading the
remaining one. Notably, BERTMap obtained the best ranking scores in all but one task, where Matcha
took the top spot, although some systems did not provide ranking results.</p>
          <p>In summary, the 2025 edition introduced four new machine learning-based systems. Although some
participants from previous years did not resubmit their tools, the growing number of learning-based
systems is consistent with Bio-ML’s original mission. Meanwhile, LogMap and its variants remained
the only symbolic systems participating in the campaign.</p>
        </sec>
        <sec id="sec-3-3-15">
          <title>4.8. Digital Humanities</title>
          <p>Among the submitted systems, Agent-OM, LogMap, LogMap-Bio, LogMapKG, Matcha and TIM
successfully found alignments. LSMatch executed without runtime errors but generated empty alignments.
ALIN and MDMapper encountered code exceptions during execution. Since Agent-OM can not be
used within the MELT framework, it could not be executed or veried by the organizers. Instead, its
alignments were provided directly by the system developers.</p>
          <p>When comparing system performance (see Table 16), Matcha achieved the highest average F1-score
of 0.64, an improvement of 0.05 over the best-performing system in the previous OAEI edition.</p>
          <p>Considering the average F1-scores across all matchers (see Table 17), values range from 0.31 to 0.61.
This variation indicates that while several systems perform reasonably well on certain test cases, there
remains substantial potential for improvement on others.</p>
          <p>Regarding runtime performance (see Table 18), most systems completed the track in under 20s.
Matcha was running about 20 times longer. LogMapKG oers the best balance between runtime and
result quality. Agent-OM did not provide runtimes due to dependency on external APIs.</p>
          <p>Overall, the number of systems capable of generating alignments increased by two compared to the
previous year. Nevertheless, several systems still failed due to runtime errors or unsupported features.
This observation is consistent with the ndings reported in our previous OM study [67], in which only
ve out of seventeen systems were able to produce valid alignments. These results highlight that many
ontology matching systems continue to face challenges when processing SKOS vocabularies. On a more
positive note, both newly participating systems this year were able to successfully handle SKOS data.
Matching system performance for the digital humanities (dh) track. The numbers are rounded to two decimal
places. The best performing matcher of each test case is highlighted.</p>
          <p>Precision</p>
          <p>Recall
Averaged evaluation metrics over all matchers for each test case of the digital humanities (dh) track.</p>
        </sec>
        <sec id="sec-3-3-16">
          <title>4.9. Archaeology Multilingual</title>
          <p>Since this track builds directly on datasets from the digital humanities track, the set of systems that
executed successfully is identical to those reported in Section 4.8.</p>
          <p>Comparing the matching systems (see Table 19), Agent-OM achieved the highest average F1-score of
0.33, outperforming last year’s best system by 0.07. TIM failed to produce alignments for almost all test
cases.</p>
          <p>Examining the F1-scores averaged across all matchers (Table 20), values range from 0.04 to 0.47. As
expected, English–English and German–German combinations are handled most eectively, while test
cases involving only non-English languages remain particularly dicult. Agent-OM shows promising
results even for these test cases and is the rst system to generate non-empty alignments for the
French–Italian test case.
Total runtime for all test cases of the digital humanities (dh) track.
Matching system performance for the archaeology multilingual track. The numbers are rounded to two decimal
places. The best performing matcher in each test case is highlighted.</p>
          <p>Precision</p>
          <p>Recall
Averaged evaluation metrics over all matchers for each test case of the archaeology multilingual track.</p>
          <p>Execution times for most systems (see Table 21) are below half a minute for the whole track, except
for Matcha and LogMapLt, with the latter producing only empty alignments.</p>
          <p>The results clearly show that handling languages other than English remains an open challenge for
ontology matching systems. Compared to last year, however, there is now one system (Agent-OM)
that explicitly addresses this issue and achieves encouraging results across several multilingual test
cases. Progress in this direction is essential, particularly for domains such as the Digital Humanities,
where research data are multilingual and research by the scholars is frequently conducted in the local
language of the respective institution.</p>
        </sec>
        <sec id="sec-3-3-17">
          <title>4.10. Circular Economy</title>
          <p>Four systems have been registered for the Circular Economy track: Agent-OM, LogMap, LogMapLt, and
Matcha. We conducted experiments by executing each system in its standard setting, and we compared
precision, F-measure, and recall. We used the MELT platform to execute our evaluations for all systems.
The only exception was Agent-OM which provided its own alignments (using commercial LLM).</p>
          <p>Table 22 shows the results for precision, F-measure, recall and the size of the alignments for the
optimal threshold. Regarding the F1-measure, LogMap achieved the best score. LogMap and Matcha
provide the correspondences with real-valued condence. Therefore, we applied thresholding during
the evaluation. Agent-OM internally applied a threshold of 0.9 before submission of alignments.</p>
          <p>LogMap and Matcha use weights to score each pair in their generated alignments. The weights in
LogMap’s alignment range from 1.0 to 0.5. There were multiple matches with the lowest weight (0.5).
These mappings were a mix of correct mappings (majority) and false positives. The computed threshold
for obtaining the highest F-measure is 0.5 (including) which corresponds to the mappings analysis.
Therefore, LogMap achieves the highest F-measure while taking all the mappings into consideration,
without threshold adjustments.</p>
          <p>In case of Matcha, the weights of its results range between 1 and 0.6. Surprisingly, all mappings with
the highest weights were false positives. The correct mapping with the lowest weight was weighted to
0.9306. This is also the computed threshold for obtaining the highest F-measure which is 0.61. Threshold
0.9306 (including) is very specic. Using a more general threshold 0.9 (including), the F-measure is just
a slightly lower 0.6. Applying the thresholds, the number of correct mappings stays the same while the
number of false positives lowers from 222 to 44 for threshold 0.9306 or 46 for threshold 0.9. Applying
the threshold 0.9, precision improves from 0.149 to 0.459, F-measure from 0.255 to 0.6, while recall
remains 0.867.</p>
          <p>Looking at the results, it can be said that when the reason an alignment was discovered was the same
name, all or at least most tools generated the mapping. LogMap and Matcha further generated some
FPs based on similar strings. Agent-OM and Matcha generated some FPs based on synonyms. All four
systems generated FPs where the same word was present in the entities names.</p>
          <p>Last year, the CE track included only one ontology pair for matching, CEON and BiOnto. Comparing
the results from last year,36 all three common participants (LogMap, LogMapLt, Matcha) signicantly
improved their performance this year (Matcha aer thresholding). Based on the false positives analysis,
just as last year, it turns out that mere string matching could be misleading, and the meaning of
entities should be better considered. This approach could be an opportunity for future performance
improvements.
36More information is provided at the results web page: https://oaei.ontologymatching.org/2025/results/ce/index.html</p>
        </sec>
        <sec id="sec-3-3-18">
          <title>4.11. Beyond Equivalence</title>
          <p>Table 23 summarises the performance of the participating matchers across all 10 datasets, 5 from
industrial classication standards (ICS) and 5 from STROMA/TaSeR (ST). Table 24 provide the results
use isAmong evaluation, which provide more ne-grained measurement for correspondences beyond
equivalence.</p>
          <p>isAmong is a novel evaluation framework designed to assess ontology matching beyond simple
equivalence. Unlike traditional approaches that focus solely on exact class-to-class matches, isAmong
introduces a relation-aware perspective by transforming alignments into sets of descendant classes—called
isAmong sets. This enables the computation of class-level Precision, Recall, and F1-Score based on the
overlap between predicted and reference descendant sets, averaged across both source and target
ontologies. By rewarding containment and partial overlap, isAmong provides a fair and ne-grained metric
even when systems do not predict the exact reference relation. Tailored for classication ontologies,
the framework avoids pre-dened weights and supports ne-grained evaluation for correspondences
with relation beyond equivalence (≡), such as subclass (≤), superclass (≥), and overlap (≃).</p>
          <p>This year we evaluated ve matchers, including LogMap, LogMap-Bio, LogMapKG, Matcha, and
MDMapper [68, 69], across 10 datasets from two families: industrial classication standards (i.e.,
ECLASS–GPC, ECLASS–UNSPSC, ETIM–ECLASS, GPC–UNSPSC, GPC–UNSPSC+), and
STROMA/TaSeR (i.e., g1-web, g2-diseases, g3-text , g5-groceries, g7-literature). We report both traditional metrics
(Precision, Recall, F1-Score), which reward only exact identicial correspondences, and isAmong metrics
(Precision*, Recall*, F1-Score*), which also give credit for partially correct relations such as superclass,
subclass and overlap.</p>
          <p>Across all 10 datasets (macro level), LogMap leads on isAmong metrics (best F1*), while LogMap-Bio
achieves the highest traditional F1-Score. For industrial classication standards, MDMapper performs
best under both evaluation regimes, particularly on isAmong (top P*, R*, F1*). Overall performance is
very low, likely due to the scarcity of true equivalences and the dominance of other relation types. This
suggests that current matchers struggle to detect relations between concepts with diering granularity
or classication perspectives. For TROMA/TaSeR, LogMap-Bio achieves the best traditional F1-Score,
while LogMap leads in Recall and all isAmong metrics (Precision*, Recall*, F1-Score*).</p>
          <p>A breakdown by dataset family provides further insight:
– Industrial Classication Standards (5 datasets): These datasets remain the most challenging.
Under isAmong evaluation, average F1* stays below 13% for all systems. MDMapper performs
best in this group (F1* ≈ 12.66%), but the overall diculty highlights the incapability of existing
tools in ontologies with structural and granularity dierences from industry cases.
– STROMA/TaSeR (5 datasets): The performance here is consistently higher. LogMap obtains
isAmong F1* ≈ 30.61%, and LogMap-Bio achieves traditional F1 ≈ 30.73%. MDMapper yields
high precision (66.15%) under the traditional metric.</p>
          <p>In summary, the rst year of the Beyond Equivalence track demonstrates that matching beyond
equivalence remains a challenging and open research problem. While relation-aware evaluation provides
a more realistic assessment of system capabilities, substantial methodological advancements are required
for high-quality alignment in practical applications such as matching product classication ontologies.</p>
        </sec>
        <sec id="sec-3-3-19">
          <title>4.12. Knowledge Graph</title>
          <p>This year we evaluated all participants with the MELT framework to include all possible submission
formats i.e., SEALS, and Web format. First, all systems are evaluated on a very small matching task37
(even those not registered for the track). This revealed that not all systems were able to handle the task,
and in the end, 6 matchers can provide results for at least one test case.</p>
          <p>Table 25 shows the results for all systems divided into class, property, instance, and overall results.
This also includes the number of tasks in which they were able to generate a non-empty alignment
(#tasks) and the average number of generated correspondences (size). We report the macro averaged
precision, F-measure, and recall results, where we do not distinguish empty and erroneous (or not
generated) alignments. The values in parentheses show the results when considering only non-empty
alignments.</p>
          <p>The resulting alignments are available for download.38 This year’s best overall system is DogMa,
which beats the baselines for the rst time (0.90 F-measure) and also achieved the highest recall (0.89).
Detailed results for each test case can be found on the OAEI results page of the track.39
37http://oaei.ontologymatching.org/2019/results/knowledgegraph/small test.zip
38http://oaei.ontologymatching.org/2025/results/knowledgegraph/knowledgegraph-alignments.zip
39http://oaei.ontologymatching.org/2025/results/knowledgegraph/index.html
00:11:37
00:11:27
00:00:00
00:56:43
64:48:07
03:31:25
01:43:25
00:34:13
00:11:37
00:11:27
00:00:00
00:56:43
64:48:07
03:31:25
01:43:25
00:34:13</p>
          <p>Property matches are still not created by all systems. LogMap and Matcha do not return any of those
mappings. One reason might be that the properties are typed as rdf:Property and not distinguished
into owl:ObjectProperty or owl:DatatypeProperty.</p>
          <p>When it comes to class matches, TIM is the overall best system with an F-measure of 0.95 (much
better than the provided baseline).</p>
          <p>For further analysis of the results, we also provide an online dashboard40 generated with MELT [70].
In this dashboard, the results can be inspected on a correspondence level. Due to the large amount of
these correspondences, it can take some time to load the full website.</p>
          <p>Regarding runtime, LSMatch (03:31:25) and LogMapLt (64:48:07) were the slowest systems. Besides
the baselines (which need around 12 minutes for all test cases) TIM (00:34:13) is the fastest system.
40http://oaei.ontologymatching.org/2025/results/knowledgegraph/knowledge graph dashboard.html</p>
        </sec>
        <sec id="sec-3-3-20">
          <title>4.13. Pharmacogenomics</title>
          <p>For this third year of the Pharmacogenomics track, no systems registered for the track. Nevertheless,
we evaluated some of the systems submitted to OAEI 2025, namely LogMap (with its dierent versions:
LogMap, LogMap-Bio, LogMapLt, and LogMapKG), LSMatch, LSMatch-Multilingual, Matcha, and TIM
using the MELT framework. LSMatch, LSMatch-Multilingual, and TIM failed to produce alignments
due to runtime errors.</p>
          <p>Regarding LogMap, similarly to last year, its dierent versions did not produce alignments between
-ary tuples; however, some versions produced alignments between other entities (e.g., components of
pharmacogenomic tuples such as drugs or genetic factors). These alignments were valid, sometimes
trivial, but are out of the scope of the Pharmacogenomics track. We link the inability of LogMap
to produce alignments between -ary tuples to the absence of labels for such tuples, as providing
labels allows all LogMap versions to produce alignments. However, when labels are present, altering
neighborhoods does not impact the produced alignments, showing that only labels are taken into
account by the dierent versions of the LogMap system. Recall that -ary tuples are reifed as abstract
entities because RDF does not allow -ary relations. Hence, labels of such reied entities are seldom
present in general, but their neighbors play a crucial role in their identity. These observations lead us
to conclude that LogMap is not adequate for the task of matching pharmacogenomic knowledge, as it
appears to rely only on labels and disregard neighbors.</p>
          <p>Matcha failed to produce alignments on the three proposed tasks due to out-of-memory errors.
However, when tested on sample tasks involving only two tuples to match, Matcha was able to produce
alignments between these -ary tuples, even in the absence of labels. We also noticed that altering the
neighborhood of tuples may have varying impacts on the produced alignments. For instance, removing
one of the two inverse relations linking one tuple to its components leads to the absence of output
alignments, whereas removing the two edges for one tuple still allows alignments to be detected. As a
result, we believe Matcha could be an interesting candidate for pharmacogenomic knowledge alignment,
even if it would require to be adapted to tackle the huge number of tuples to align, and the diversity of
their neighborhoods.</p>
        </sec>
      </sec>
    </sec>
    <sec id="sec-4">
      <title>5. Conclusions and Lessons Learned</title>
      <p>As in previous campaigns, we witnessed a healthy mix of new and returning systems, with an imbalanced
participation in the tracks.</p>
      <p>The schema matching tracks gather the highest number of participants; however still little
substantial progress in terms of the quality of the results or runtime of top matching systems. As already
reported in the last years, we observe a performance plateau being reached by existing strategies and
algorithms. It is also true that established matching systems tend to focus more on new tracks and
datasets than on improving their performance in long-standing tracks, whereas new systems typically
struggle to compete with established ones.</p>
      <p>With respect to the cross-lingual version of the Conference, the Multifarm track still attracts too
few participants. Despite this fact, this year, new participants came up with alternative strategies (i.e.,
deep learning) with respect to the last campaigns.</p>
      <p>The Bio-ML track attracted several new machine learning-based participants. However, the number
of symbolic participants is still low. The best-performing systems are not consistent across tasks and
settings, demonstrating the diversity of our datasets.</p>
      <p>The results of the Digital Humanities track show that SKOS vocabularies are still not well-supported
by many matching systems. However, there is an improvement to last year with two new systems
capable of handling SKOS. For all systems, there is still room for improvement.</p>
      <p>The Archaeology multilingual track leads to the conclusion that languages other than English
are not well-supported. The newcomer Agent-OM showed promising results in this track. In future
versions, it is planned to include other languages such as Japanese.</p>
      <p>The Interactive matching track also witnessed a small number of participants. Two systems
participated this year. This is puzzling considering that this track is based on the Anatomy and Conference
test cases, and those tracks had 11 and 7 participants, respectively. The process of programmatically
querying the Oracle class used to simulate user interactions is simple enough that it should not be a
deterrent for participation, but perhaps we should look at facilitating the process further in future OAEI
editions by providing implementation examples.</p>
      <p>The Complex matching track tackles a challenge task that attracts too few number of participants.
This year, two systems were able to complete the task. A new dataset is an addition this year for the
specic task of complex multi-ontology matching.</p>
      <p>Automatic instance-matching benchmark generation algorithms have been gaining popularity, as
evidenced by the fact that they are used in instance-matching tracks. One aspect that has not been
addressed in such algorithms is that, if the transformation is too extreme, the correspondence may
be unrealistic and impossible to detect even by humans. As such, we argue that human-in-the-loop
techniques can be exploited to do a preventive quality-checking of generated correspondences and
rene the set of correspondences included in the nal reference alignment.</p>
      <p>In the Knowledge graph track, we have two new matching systems, TIM and DogMa, which beat
the previous best F-Measures in class/property and instance performance. We hope that, in the future,
more systems will focus on the track and continue to improve.</p>
      <p>For the third year of the Pharmacogenomics track, we tested several systems submitted to OAEI
2025, even if they did not specically register for the track. None of these systems were successful in
producing alignments between reied -ary tuples, which, according to our investigation, is due to the
absence of labels for the -ary tuples to align, or to scalability issues. In particular, Matcha is the most
promising candidate as it is able to match -ary tuples but fails due to their high number. These results
highlight the interest of considering domain-specic problems that bring additional challenges to the
eld of ontology matching (here, dierent types of alignments between individuals, structure-based
matching). Given the inability of registered systems to produce valid alignments, such challenges are
currently unaddressed and require to design new methods like [58, 71] or enrich existing ones. This
ultimately motivates to propose the track again in future editions of OAEI, hoping to motivate new
systems targeting this real-world matching scenario. We will also enrich the track with tasks that
involve fewer entities to match, or specic variations in tuple neighborhoods, allowing to assess even
further systems capabilities and limits.</p>
      <p>Like in previous OAEI editions, most participants provided a description of their systems and their
experience in the evaluation, in the form of OAEI system papers. These papers, like the present one,
have not been peer-reviewed. However, they are full contributions to this evaluation exercise, reecting
the eort and insight of matching systems developers, and providing details about those systems and
the algorithms they implement.</p>
      <p>As each year, fruitful discussions at the Ontology Matching Workshop point out dierent directions
for future improvements in OAEI. Since 2023, as a growing number of systems rely on Large Language
Models, we started discussing the specic requirements and alternative ways of gathering the alignments
generated by such resource-consuming systems. This year, it is highlighted that more systems rely on
external calls to LLMs, which are not yet well adapted to run within our evaluation platforms. Hence, a
platform capable of running resource-intensive systems is needed, potentially supported by funding or
an infrastructure project, while complex alignments and other relations than equivalence remain to be
addressed, reference alignments in the OAEI datasets should be revised, and an in-use session featuring
real industrial cases (possibly as a special workshop session) would be valuable.</p>
      <p>The Ontology Alignment Evaluation Initiative will strive to remain a reference to the ontology
matching community by improving both the test cases and the testing methodology to better reect
actual needs, as well as to promote progress in this eld. More information can be found at: http:
//oaei.ontologymatching.org.</p>
    </sec>
    <sec id="sec-5">
      <title>Acknowledgments</title>
      <p>We warmly thank the participants of this campaign. We know that they have worked hard to have their
matching tools executable in time and they provided useful reports on their experience. The best way
to learn about the results remains to read the papers that follow.</p>
      <p>We are also grateful to Martin Ringwald and Terry Hayamizu for providing the reference alignment
for the anatomy ontologies and thank Elena Beisswanger for her thorough support in improving the
dataset’s quality.</p>
      <p>We also thank for their support, the past members of the Ontology Alignment Evaluation Initiative
steering committee: Je´ro^me Euzenat (INRIA, FR), Yannis Kalfoglou (Ricoh laboratories, UK), Miklos
Nagy (The Open University, UK), Natasha Noy (Google Inc., USA), Yuzhong Qu (Southeast University,
CN), York Sure (Leibniz Gemeinscha, DE), Jie Tang (Tsinghua University, CN), Heiner Stuckenschmidt
(Mannheim Universita¨t, DE), and George Vouros (University of the Aegean, GR).</p>
      <p>Catia Pesquita was supported by the FCT through the LASIGE Research Unit (UIDB/00408/2020 and
UIDP/00408/2020) and by the KATY project funded by the European Union’s Horizon 2020 research
and innovation program under grant agreement No 101017453.</p>
      <p>Patrick Lambrix, Mina Abd Nikooie Pour and Ying Li have been supported by the Swedish e-Science
Research Centre (SeRC) and the Swedish National Graduate School in Computer Science (CUGS).</p>
      <p>Eva Blomqvist, Patrick Lambrix, Huanyu Li, Ondrˇej Zamazal and Jana Vatasˇcˇinova´ have been
supported by the European Union’s Horizon Europe research and innovation programme under grant
agreement no. 101058682 (Onto-DESIDE).</p>
      <p>Beyza Yaman has been supported by ADAPT SFI Research Centre [grant 13/RC/2106 P2].</p>
      <p>The work of Felix Kraus was funded by the research program “Engineering Digital Futures” of
the Helmholtz Association of German Research Centers, and the Helmholtz Metadata Collaboration
Platform (HMC).</p>
    </sec>
    <sec id="sec-6">
      <title>Declaration on Generative AI</title>
      <sec id="sec-6-1">
        <title>The author(s) have not employed any Generative AI tools.</title>
        <p>(ISWC 2024), Baltimore, USA, November 11, 2024, volume 3897 of CEUR Workshop Proceedings,
CEUR-WS.org, 2024, pp. 64–97. URL: https://ceur-ws.org/Vol-3897/oaei2024 paper0.pdf.
[6] M. Abd Nikooie Pour, A. Algergawy, P. Buche, L. J. Castro, J. Chen, A. Coulet, J. Cu, H. Dong,
O. Fallatah, D. Faria, I. Fundulaki, S. Hertling, Y. He, I. Horrocks, M. Huschka, L. Ibanescu, S. Jain,
E. Jime´nez-Ruiz, N. Karam, P. Lambrix, H. Li, Y. Li, P. Monnin, E. Nasr, H. Paulheim, C. Pesquita,
T. Saveta, P. Shvaiko, G. Sousa, C. Trojahn, J. Vatascinova, M. Wu, B. Yaman, O. Zamazal, L. Zhou,
Results of the Ontology Alignment Evaluation Initiative 2023, in: P. Shvaiko, J. Euzenat, E.
Jime´nezRuiz, O. Hassanzadeh, C. Trojahn (Eds.), Proceedings of the 18th International Workshop on
Ontology Matching (OM 2023) co-located with the 22nd International Semantic Web Conference
(ISWC 2023), Athens, Greece, November 7, 2023, volume 3591 of CEUR Workshop Proceedings,
CEUR-WS.org, 2023, pp. 97–139. URL: https://ceur-ws.org/Vol-3591/oaei23 paper0.pdf.
[7] M. Abd Nikooie Pour, A. Algergawy, P. Buche, L. J. Castro, J. Chen, H. Dong, O. Fallatah, D. Faria,
I. Fundulaki, S. Hertling, Y. He, I. Horrocks, M. Huschka, L. Ibanescu, E. Jime´nez-Ruiz, N. Karam,
A. Laadhar, P. Lambrix, H. Li, Y. Li, F. Michel, E. Nasr, H. Paulheim, C. Pesquita, T. Saveta, P. Shvaiko,
C. Trojahn, C. Verhey, M. Wu, B. Yaman, O. Zamazal, L. Zhou, Results of the Ontology Alignment
Evaluation Initiative 2022, in: P. Shvaiko, J. Euzenat, E. Jime´nez-Ruiz, O. Hassanzadeh, C. Trojahn
(Eds.), Proceedings of the 17th International Workshop on Ontology Matching (OM 2022) co-located
with the 21th International Semantic Web Conference (ISWC 2022), Hangzhou, China, held as a
virtual conference, October 23, 2022, volume 3324 of CEUR Workshop Proceedings, CEUR-WS.org,
2022, pp. 84–128. URL: https://ceur-ws.org/Vol-3324/oaei22 paper0.pdf.
[8] M. Abd Nikooie Pour, A. Algergawy, F. Amardeilh, R. Amini, O. Fallatah, D. Faria, I. Fundulaki,
I. Harrow, S. Hertling, P. Hitzler, M. Huschka, L. Ibanescu, E. Jime´nez-Ruiz, N. Karam, A. Laadhar,
P. Lambrix, H. Li, Y. Li, F. Michel, E. Nasr, H. Paulheim, C. Pesquita, J. Portisch, C. Roussey, T. Saveta,
P. Shvaiko, A. Splendiani, C. Trojahn, J. Vatascinova´, B. Yaman, O. Zamazal, L. Zhou, Results of
the Ontology Alignment Evaluation Initiative 2021, in: P. Shvaiko, J. Euzenat, E. Jime´nez-Ruiz,
O. Hassanzadeh, C. Trojahn (Eds.), Proceedings of the 16th International Workshop on Ontology
Matching co-located with the 20th International Semantic Web Conference (ISWC 2021), Virtual
conference, October 25, 2021, volume 3063 of CEUR Workshop Proceedings, CEUR-WS.org, 2021,
pp. 62–108. URL: http://ceur-ws.org/Vol-3063/oaei21 paper0.pdf.
[9] M. Abd Nikooie Pour, A. Algergawy, R. Amini, D. Faria, I. Fundulaki, I. Harrow, S. Hertling,
E. Jime´nez-Ruiz, C. Jonquet, N. Karam, A. Khiat, A. Laadhar, P. Lambrix, H. Li, Y. Li, P. Hitzler,
H. Paulheim, C. Pesquita, T. Saveta, P. Shvaiko, A. Splendiani, E´ . Thie´blin, C. Trojahn, J. Vatascinova´,
B. Yaman, O. Zamazal, L. Zhou, Results of the Ontology Alignment Evaluation Initiative 2020, in:
P. Shvaiko, J. Euzenat, E. Jime´nez-Ruiz, O. Hassanzadeh, C. Trojahn (Eds.), Proceedings of the 15th
International Workshop on Ontology Matching co-located with the 19th International Semantic
Web Conference (ISWC 2020), Virtual conference (originally planned to be in Athens, Greece),
November 2, 2020, volume 2788 of CEUR Workshop Proceedings, CEUR-WS.org, 2020, pp. 92–138.</p>
        <p>URL: http://ceur-ws.org/Vol-2788/oaei20 paper0.pdf.
[10] A. Algergawy, D. Faria, A. Ferrara, I. Fundulaki, I. Harrow, S. Hertling, E. Jime´nez-Ruiz, N. Karam,
A. Khiat, P. Lambrix, H. Li, S. Montanelli, H. Paulheim, C. Pesquita, T. Saveta, P. Shvaiko, A.
Splendiani, E´ . Thie´blin, C. Trojahn, J. Vatascinova´, O. Zamazal, L. Zhou, Results of the Ontology Alignment
Evaluation Initiative 2019, in: Proceedings of the 14th International Workshop on Ontology
Matching, Auckland, New Zealand, volume 2536 of CEUR Workshop Proceedings, CEUR-WS.org, 2019, pp.
46–85. URL: https://ceur-ws.org/Vol-2536/oaei19 paper0.pdf.
[11] A. Algergawy, M. Cheatham, D. Faria, A. Ferrara, I. Fundulaki, I. Harrow, S. Hertling, E.
Jime´nezRuiz, N. Karam, A. Khiat, P. Lambrix, H. Li, S. Montanelli, H. Paulheim, C. Pesquita, T. Saveta,
D. Schmidt, P. Shvaiko, A. Splendiani, E´ . Thie´blin, C. Trojahn, J. Vatascinova´, O. Zamazal, L. Zhou,
Results of the Ontology Alignment Evaluation Initiative 2018, in: Proceedings of the 13th
International Workshop on Ontology Matching, Monterey (CA, US), volume 2288 of CEUR Workshop
Proceedings, CEUR-WS.org, 2018, pp. 76–116. URL: https://ceur-ws.org/Vol-2288/oaei18 paper0.pdf.
[12] M. Achichi, M. Cheatham, Z. Dragisic, J. Euzenat, D. Faria, A. Ferrara, G. Flouris, I. Fundulaki, I.
Harrow, V. Ivanova, E. Jime´nez-Ruiz, K. Koltho, E. Kuss, P. Lambrix, H. Leopold, H. Li, C. Meilicke,
M. Mohammadi, S. Montanelli, C. Pesquita, T. Saveta, P. Shvaiko, A. Splendiani, H. Stuckenschmidt,
E´. Thie´blin, K. Todorov, C. Trojahn, O. Zamazal, Results of the Ontology Alignment Evaluation
Initiative 2017, in: Proceedings of the 12th International Workshop on Ontology Matching,
Vienna, Austria, volume 2032 of CEUR Workshop Proceedings, CEUR-WS.org, 2017, pp. 61–113. URL:
http://ceur-ws.org/Vol-2032/oaei17 paper0.pdf.
[13] M. Achichi, M. Cheatham, Z. Dragisic, J. Euzenat, D. Faria, A. Ferrara, G. Flouris, I. Fundulaki,
I. Harrow, V. Ivanova, E. Jime´nez-Ruiz, E. Kuss, P. Lambrix, H. Leopold, H. Li, C. Meilicke, S.
Montanelli, C. Pesquita, T. Saveta, P. Shvaiko, A. Splendiani, H. Stuckenschmidt, K. Todorov, C. Trojahn,
O. Zamazal, Results of the Ontology Alignment Evaluation Initiative 2016, in: Proceedings of
the 11th International Workshop on Ontology Matching co-located with the 15th International
Semantic Web Conference (ISWC 2016), Kobe, Japan, volume 1766 of CEUR Workshop Proceedings,
CEUR-WS.org, 2016, pp. 73–129. URL: https://ceur-ws.org/Vol-1766/oaei16 paper0.pdf.
[14] M. Cheatham, Z. Dragisic, J. Euzenat, D. Faria, A. Ferrara, G. Flouris, I. Fundulaki, R. Granada,
V. Ivanova, E. Jime´nez-Ruiz, P. Lambrix, S. Montanelli, C. Pesquita, T. Saveta, P. Shvaiko,
A. Solimando, C. Trojahn, O. Zamazal, Results of the Ontology Alignment Evaluation
Initiative 2015, in: Proceedings of the 10th International Workshop on Ontology Matching
collocated with the 14th International Semantic Web Conference (ISWC 2015), Bethlehem,
PA, USA, volume 1545 of CEUR Workshop Proceedings, CEUR-WS.org, 2015, pp. 60–115. URL:
https://ceur-ws.org/Vol-1545/oaei15 paper0.pdf.
[15] Z. Dragisic, K. Eckert, J. Euzenat, D. Faria, A. Ferrara, R. Granada, V. Ivanova, E. Jime´nez-Ruiz,
A. O. Kempf, P. Lambrix, S. Montanelli, H. Paulheim, D. Ritze, P. Shvaiko, A. Solimando, C. T. dos
Santos, O. Zamazal, B. C. Grau, Results of the Ontology Alignment Evaluation Initiative 2014,
in: Proceedings of the 9th International Workshop on Ontology Matching collocated with the
13th International Semantic Web Conference (ISWC 2014), Riva del Garda (IT), volume 1317 of
CEUR Workshop Proceedings, CEUR-WS.org, 2014, pp. 61–104. URL: http://ceur-ws.org/Vol-1317/
oaei14 paper0.pdf.
[16] B. Cuenca Grau, Z. Dragisic, K. Eckert, J. Euzenat, A. Ferrara, R. Granada, V. Ivanova, E.
Jime´nezRuiz, A. Kempf, P. Lambrix, A. Nikolov, H. Paulheim, D. Ritze, F. Schare, P. Shvaiko, C. Trojahn dos
Santos, O. Zamazal, Results of the Ontology Alignment Evaluation Initiative 2013, in: P. Shvaiko,
J. Euzenat, K. Srinivas, M. Mao, E. Jime´nez-Ruiz (Eds.), Proceedings of the 8th International
Workshop on Ontology Matching co-located with the 12th International Semantic Web Conference
(ISWC 2013), Sydney (NSW, AU), volume 1111 of CEUR Workshop Proceedings, CEUR-WS.org, 2013,
pp. 61–100. URL: https://ceur-ws.org/Vol-1111/oaei13 paper0.pdf.
[17] J. Aguirre, B. Cuenca Grau, K. Eckert, J. Euzenat, A. Ferrara, R. van Hague, L. Hollink, E.
Jime´nezRuiz, C. Meilicke, A. Nikolov, D. Ritze, F. Schare, P. Shvaiko, O. Sva´b-Zamazal, C. Trojahn,
B. Zapilko, Results of the Ontology Alignment Evaluation Initiative 2012, in: Proceedings
of the 7th International Workshop on Ontology Matching (OM-2012) collocated with the 11th
International Semantic Web Conference (ISWC-2012), Boston (MA, US), volume 946 of CEUR
Workshop Proceedings, CEUR-WS.org, 2012, pp. 73–115. URL: https://ceur-ws.org/Vol-946/oaei12
paper0.pdf.
[18] J. Euzenat, A. Ferrara, R. van Hague, L. Hollink, C. Meilicke, A. Nikolov, F. Schare, P. Shvaiko,
H. Stuckenschmidt, O. Sva´b-Zamazal, C. Trojahn dos Santos, Results of the Ontology Alignment
Evaluation Initiative 2011, in: Proceedings of the 6th International Workshop on Ontology
Matching, Bonn (DE), volume 814 of CEUR Workshop Proceedings, CEUR-WS.org, 2011, pp. 85–110.</p>
        <p>URL: https://ceur-ws.org/Vol-814/oaei11 paper0.pdf.
[19] J. Euzenat, A. Ferrara, C. Meilicke, A. Nikolov, J. Pane, F. Schare, P. Shvaiko, H. Stuckenschmidt,
O. Sva´b-Zamazal, V. Sva´tek, C. Trojahn dos Santos, Results of the Ontology Alignment Evaluation
Initiative 2010, in: Proceedings of the 5th International Workshop on Ontology Matching
(OM2010) collocated with the 9th International Semantic Web Conference (ISWC-2010), Shanghai
(CN), volume 689 of CEUR Workshop Proceedings, CEUR-WS.org, 2010, pp. 85–117. URL: https:
//ceur-ws.org/Vol-689/oaei10 paper0.pdf.
[20] J. Euzenat, A. Ferrara, L. Hollink, A. Isaac, C. Joslyn, V. Malaise´, C. Meilicke, A. Nikolov, J. Pane,
[63] Z. Qiang, W. Wang, K. Taylor, Agent-OM: Leveraging LLM Agents for Ontology Matching, Proc.</p>
        <p>VLDB Endow. 18 (2024) 516–529. doi:10.14778/3712221.3712222.
[64] S. Oulei, L. Berkani, N. Boudjenah, L. Bellatreche, A. Mokhtari, BioGITOM: Matching Biomedical
Ontologies with Graph Isomorphism Transformer, The VLDB Journal 34 (2025) 65. doi:10.1007/
s00778-025-00943-7.
[65] D. Faria, M. C. Silva, P. Cotovio, P. Euge´nio, C. Pesquita, Matcha and Matcha-DL results for
OAEI 2022, in: Proceedings of the 17th International Workshop on Ontology Matching (OM
2022) co-located with the 21th International Semantic Web Conference (ISWC 2022), Hangzhou,
China, held as a virtual conference, October 23, 2022, volume 3324 of CEUR Workshop Proceedings,
CEUR-WS.org, 2022, pp. 197–201. URL: https://ceur-ws.org/Vol-3324/oaei22 paper11.pdf.
[66] D. Faria, M. C. Silva, P. Cotovio, L. Ferraz, L. Balbi, C. Pesquita, Results for Matcha and Matcha-DL
in OAEI 2023, in: Proceedings of the 18th International Workshop on Ontology Matching (OM
2023) co-located with the 22nd International Semantic Web Conference (ISWC 2023), Athens,
Greece, November 7, 2023, volume 3591 of CEUR Workshop Proceedings, CEUR-WS.org, 2023, pp.
164–169. URL: https://ceur-ws.org/Vol-3591/oaei23 paper6.pdf.
[67] F. Kraus, N. Blumenro¨hr, G. Go¨tzelmann, D. Tonne, A. Streit, A Gold Standard Benchmark Dataset
for Digital Humanities, in: Proceedings of the 19th International Workshop on Ontology Matching
co-located with the 23rd International Semantic Web Conference (ISWC 2024), Baltimore, USA,
November 11th, 2024, volume 3897 of CEUR Workshop Proceedings, CEUR-WS.org, 2024. URL:
https://ceur-ws.org/Vol-3897/om2024 LTpaper1.pdf.
[68] X. Liu, J. Grode, M. R. Hansen, MDMapper: A Framework for Aligning Master Data Models
using Ontology Matching Techniques, in: Proceedings of the 19th International Workshop on
Ontology Matching co-located with the 23rd International Semantic Web Conference (ISWC 2024),
Baltimore, USA, November 11th, 2024, volume 3897 of CEUR Workshop Proceedings, CEUR-WS.org,
2024. URL: https://ceur-ws.org/Vol-3897/om2024 LTpaper3.pdf.
[69] X. Liu, M. R. Hansen, J. Grode, MDMapper Results for OAEI 2024, in: Proceedings of the 19th
International Workshop on Ontology Matching co-located with the 23rd International Semantic
Web Conference (ISWC 2024), Baltimore, USA, November 11th, 2024, volume 3897 of CEUR
Workshop Proceedings, CEUR-WS.org, 2024. URL: https://ceur-ws.org/Vol-3897/oaei2024 paper1.
pdf.
[70] J. Portisch, S. Hertling, H. Paulheim, Visual Analysis of Ontology Matching Results with the MELT
Dashboard, in: The Semantic Web: ESWC 2020 Satellite Events, 2020, pp. 186–190. doi:10.1007/
978-3-030-62327-2 32.
[71] P. Monnin, C. Ra¨ıssi, A. Napoli, A. Coulet, Discovering alignment relations with Graph
Convolutional Networks: A biomedical case study, Semantic Web 13 (2022) 379–398. doi:10.3233/
SW-210452.</p>
      </sec>
    </sec>
  </body>
  <back>
    <ref-list>
      <ref id="ref1">
        <mixed-citation>
          [1]
          <string-name>
            <given-names>J.</given-names>
            <surname>Euzenat</surname>
          </string-name>
          ,
          <string-name>
            <given-names>C.</given-names>
            <surname>Meilicke</surname>
          </string-name>
          ,
          <string-name>
            <given-names>P.</given-names>
            <surname>Shvaiko</surname>
          </string-name>
          ,
          <string-name>
            <given-names>H.</given-names>
            <surname>Stuckenschmidt</surname>
          </string-name>
          ,
          <string-name>
            <surname>C.</surname>
          </string-name>
          <article-title>Trojahn dos Santos, Ontology Alignment Evaluation Initiative: Six Years of Experience, Journal on Data Semantics XV (</article-title>
          <year>2011</year>
          )
          <fpage>158</fpage>
          -
          <lpage>192</lpage>
          . doi:
          <volume>10</volume>
          .1007/978-3-
          <fpage>642</fpage>
          -22630-4
          <fpage>6</fpage>
          .
        </mixed-citation>
      </ref>
      <ref id="ref2">
        <mixed-citation>
          [2]
          <string-name>
            <given-names>J.</given-names>
            <surname>Euzenat</surname>
          </string-name>
          ,
          <string-name>
            <given-names>P.</given-names>
            <surname>Shvaiko</surname>
          </string-name>
          , Ontology Matching, second edition ed., Springer, Berlin, Heidelberg,
          <year>2013</year>
          . doi:
          <volume>10</volume>
          .1007/978-3-
          <fpage>642</fpage>
          -38721-0.
        </mixed-citation>
      </ref>
      <ref id="ref3">
        <mixed-citation>
          [3]
          <string-name>
            <given-names>Y.</given-names>
            <surname>Sure</surname>
          </string-name>
          ,
          <string-name>
            <given-names>O.</given-names>
            <surname>Corcho</surname>
          </string-name>
          ,
          <string-name>
            <given-names>J.</given-names>
            <surname>Euzenat</surname>
          </string-name>
          , T. Hughes (Eds.),
          <source>Proceedings of the 3rd International Workshop on Evaluation of Ontology-based Tools held at the 3rd International Semantic Web Conference ISWC</source>
          <year>2004</year>
          , Hiroshima, Japan, volume
          <volume>128</volume>
          ,
          <year>2004</year>
          . URL: https://ceur-ws.
          <source>org/</source>
          Vol-
          <volume>128</volume>
          /.
        </mixed-citation>
      </ref>
      <ref id="ref4">
        <mixed-citation>
          [4]
          <string-name>
            <given-names>B.</given-names>
            <surname>Ashpole</surname>
          </string-name>
          ,
          <string-name>
            <given-names>M.</given-names>
            <surname>Ehrig</surname>
          </string-name>
          ,
          <string-name>
            <given-names>J.</given-names>
            <surname>Euzenat</surname>
          </string-name>
          , H. Stuckenschmidt (Eds.),
          <source>Proceedings of the K-CAP 2005 Workshop on Integrating Ontologies</source>
          , volume
          <volume>156</volume>
          ,
          <string-name>
            <surname>Ban</surname>
            <given-names></given-names>
          </string-name>
          (Canada),
          <year>2005</year>
          . URL: http://ceur-ws.
          <source>org/</source>
          Vol-
          <volume>156</volume>
          /.
        </mixed-citation>
      </ref>
      <ref id="ref5">
        <mixed-citation>
          [5]
          <string-name>
            <given-names>M.</given-names>
            <surname>Abd Nikooie Pour</surname>
          </string-name>
          ,
          <string-name>
            <given-names>A.</given-names>
            <surname>Algergawy</surname>
          </string-name>
          , E. Blomqvist,
          <string-name>
            <given-names>P.</given-names>
            <surname>Buche</surname>
          </string-name>
          ,
          <string-name>
            <given-names>J.</given-names>
            <surname>Chen</surname>
          </string-name>
          ,
          <string-name>
            <given-names>A.</given-names>
            <surname>Coulet</surname>
          </string-name>
          , J. Cu,
          <string-name>
            <given-names>H.</given-names>
            <surname>Dong</surname>
          </string-name>
          ,
          <string-name>
            <given-names>D.</given-names>
            <surname>Faria</surname>
          </string-name>
          ,
          <string-name>
            <given-names>L.</given-names>
            <surname>Ferraz</surname>
          </string-name>
          ,
          <string-name>
            <given-names>P. G.</given-names>
            <surname>Cotovio</surname>
          </string-name>
          ,
          <string-name>
            <given-names>Y.</given-names>
            <surname>He</surname>
          </string-name>
          ,
          <string-name>
            <given-names>S.</given-names>
            <surname>Hertling</surname>
          </string-name>
          ,
          <string-name>
            <surname>I. Horrocks</surname>
          </string-name>
          ,
          <string-name>
            <given-names>L.</given-names>
            <surname>Ibanescu</surname>
          </string-name>
          ,
          <string-name>
            <given-names>S.</given-names>
            <surname>Jain</surname>
          </string-name>
          , E. Jime´nezRuiz,
          <string-name>
            <given-names>N.</given-names>
            <surname>Karam</surname>
          </string-name>
          ,
          <string-name>
            <given-names>F.</given-names>
            <surname>Kraus</surname>
          </string-name>
          ,
          <string-name>
            <given-names>P.</given-names>
            <surname>Lambrix</surname>
          </string-name>
          ,
          <string-name>
            <given-names>H.</given-names>
            <surname>Li</surname>
          </string-name>
          ,
          <string-name>
            <given-names>Y.</given-names>
            <surname>Li</surname>
          </string-name>
          ,
          <string-name>
            <given-names>P.</given-names>
            <surname>Monnin</surname>
          </string-name>
          ,
          <string-name>
            <given-names>H.</given-names>
            <surname>Paulheim</surname>
          </string-name>
          ,
          <string-name>
            <given-names>C.</given-names>
            <surname>Pesquita</surname>
          </string-name>
          ,
          <string-name>
            <given-names>A.</given-names>
            <surname>Sharma</surname>
          </string-name>
          ,
          <string-name>
            <given-names>P.</given-names>
            <surname>Shvaiko</surname>
          </string-name>
          ,
          <string-name>
            <given-names>M.</given-names>
            <surname>Silva</surname>
          </string-name>
          , G. Sousa,
          <string-name>
            <given-names>C.</given-names>
            <surname>Trojahn</surname>
          </string-name>
          ,
          <string-name>
            <given-names>J.</given-names>
            <surname>Vatascinova</surname>
          </string-name>
          ,
          <string-name>
            <given-names>B.</given-names>
            <surname>Yaman</surname>
          </string-name>
          ,
          <string-name>
            <given-names>O.</given-names>
            <surname>Zamazal</surname>
          </string-name>
          ,
          <string-name>
            <given-names>L.</given-names>
            <surname>Zhou</surname>
          </string-name>
          ,
          <source>Results of the Ontology Alignment Evaluation Initiative</source>
          <year>2024</year>
          , in: E.
          <string-name>
            <surname>Jime</surname>
          </string-name>
          <article-title>´nez-</article-title>
          <string-name>
            <surname>Ruiz</surname>
            ,
            <given-names>O.</given-names>
          </string-name>
          <string-name>
            <surname>Hassanzadeh</surname>
            ,
            <given-names>C.</given-names>
          </string-name>
          <string-name>
            <surname>Trojahn</surname>
            ,
            <given-names>S.</given-names>
          </string-name>
          <string-name>
            <surname>Hertling</surname>
            ,
            <given-names>H.</given-names>
          </string-name>
          <string-name>
            <surname>Li</surname>
            ,
            <given-names>P.</given-names>
          </string-name>
          <string-name>
            <surname>Shvaiko</surname>
          </string-name>
          , J. Euzenat (Eds.),
          <source>Proceedings of the 19th International Workshop on Ontology Matching co-located with the 23rd International Semantic Web Conference</source>
        </mixed-citation>
      </ref>
    </ref-list>
  </back>
</article>