<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD v1.0 20120330//EN" "JATS-archivearticle1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink">
  <front>
    <journal-meta>
      <journal-title-group>
        <journal-title>and Patrick Lambrix. Experiences from
the anatomy track in the ontology alignment evaluation initiative. Journal of Biomedical
Semantics</journal-title>
      </journal-title-group>
    </journal-meta>
    <article-meta>
      <title-group>
        <article-title>Results of the Ontology Alignment Evaluation Initiative 2020?</article-title>
      </title-group>
      <contrib-group>
        <contrib contrib-type="author">
          <string-name>Mina Abd Nikooie Pour</string-name>
          <xref ref-type="aff" rid="aff10">10</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Alsayed Algergawy</string-name>
          <email>alsayed.algergawy@uni-jena.de</email>
          <xref ref-type="aff" rid="aff5">5</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Reihaneh Amini</string-name>
          <xref ref-type="aff" rid="aff2">2</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Daniel Faria</string-name>
          <email>dfaria@inesc-id.pt</email>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Irini Fundulaki</string-name>
          <xref ref-type="aff" rid="aff7">7</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Ian Harrow</string-name>
          <xref ref-type="aff" rid="aff12">12</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Sven Hertling</string-name>
          <xref ref-type="aff" rid="aff15">15</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Ernesto Jime´nez-Ruiz</string-name>
          <email>ernesto.jimenez-ruiz@city.ac.uk</email>
          <email>ernestoj@ifi.uio.no</email>
          <xref ref-type="aff" rid="aff1">1</xref>
          <xref ref-type="aff" rid="aff3">3</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Clement Jonquet</string-name>
          <xref ref-type="aff" rid="aff9">9</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Naouel Karam</string-name>
          <email>naouel.karam@fokus.fraunhofer.de</email>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Abderrahmane Khiat</string-name>
          <xref ref-type="aff" rid="aff4">4</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Amir Laadhar</string-name>
          <xref ref-type="aff" rid="aff9">9</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Patrick Lambrix</string-name>
          <xref ref-type="aff" rid="aff10">10</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Huanyu Li</string-name>
          <xref ref-type="aff" rid="aff10">10</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Ying Li</string-name>
          <email>ying.lig@liu.se</email>
          <xref ref-type="aff" rid="aff10">10</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Pascal Hitzler</string-name>
          <xref ref-type="aff" rid="aff2">2</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Heiko Paulheim</string-name>
          <email>heikog@informatik.uni-mannheim.de</email>
          <xref ref-type="aff" rid="aff15">15</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Catia Pesquita</string-name>
          <email>cpesquita@di.fc.ul.pt</email>
          <xref ref-type="aff" rid="aff8">8</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Tzanina Saveta</string-name>
          <xref ref-type="aff" rid="aff7">7</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Pavel Shvaiko</string-name>
          <email>pavel.shvaiko@tndigit.it</email>
          <xref ref-type="aff" rid="aff13">13</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Andrea Splendiani</string-name>
          <xref ref-type="aff" rid="aff12">12</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Elodie Thie´blin</string-name>
          <email>elodie.thieblin@logilab.fr</email>
          <xref ref-type="aff" rid="aff11">11</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Ca´ssia Trojahn</string-name>
          <email>cassia.trojahn@irit.fr</email>
          <xref ref-type="aff" rid="aff6">6</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Jana Vatasˇcˇinova´</string-name>
          <xref ref-type="aff" rid="aff14">14</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Beyza Yaman</string-name>
          <xref ref-type="aff" rid="aff0">0</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Ondrˇej Zamazal</string-name>
          <email>ondrej.zamazalg@vse.cz</email>
          <xref ref-type="aff" rid="aff14">14</xref>
        </contrib>
        <contrib contrib-type="author">
          <string-name>Lu Zhou</string-name>
          <xref ref-type="aff" rid="aff2">2</xref>
        </contrib>
        <aff id="aff0">
          <label>0</label>
          <institution>ADAPT Centre, Dublin City University</institution>
          ,
          <addr-line>Ireland beyza.yamanadaptcentre.ie</addr-line>
        </aff>
        <aff id="aff1">
          <label>1</label>
          <institution>City, University of London</institution>
          ,
          <country country="UK">UK</country>
        </aff>
        <aff id="aff2">
          <label>2</label>
          <institution>Data Semantics (DaSe) Laboratory, Kansas State University</institution>
          ,
          <country country="US">USA</country>
        </aff>
        <aff id="aff3">
          <label>3</label>
          <institution>Department of Informatics, University of Oslo</institution>
          ,
          <country country="NO">Norway</country>
        </aff>
        <aff id="aff4">
          <label>4</label>
          <institution>Fraunhofer IAIS</institution>
          ,
          <addr-line>Sankt Augustin</addr-line>
          ,
          <country country="DE">Germany</country>
        </aff>
        <aff id="aff5">
          <label>5</label>
          <institution>Friedrich Schiller University Jena</institution>
          ,
          <country country="DE">Germany</country>
        </aff>
        <aff id="aff6">
          <label>6</label>
          <institution>IRIT &amp; Universite ́ Toulouse II</institution>
          ,
          <addr-line>Toulouse</addr-line>
          ,
          <country country="FR">France</country>
        </aff>
        <aff id="aff7">
          <label>7</label>
          <institution>Institute of Computer Science-FORTH</institution>
          ,
          <addr-line>Heraklion</addr-line>
          ,
          <country country="GR">Greece</country>
        </aff>
        <aff id="aff8">
          <label>8</label>
          <institution>LASIGE, Faculdade de Cieˆncias, Universidade de Lisboa</institution>
          ,
          <country country="PT">Portugal</country>
        </aff>
        <aff id="aff9">
          <label>9</label>
          <institution>LIRMM, University of Montpellier &amp; CNRS</institution>
          ,
          <country country="FR">France</country>
        </aff>
        <aff id="aff10">
          <label>10</label>
          <institution>Linko ̈ping University &amp; Swedish e-Science Research Center</institution>
          ,
          <addr-line>Linko ̈ping</addr-line>
          ,
          <country country="SE">Sweden</country>
        </aff>
        <aff id="aff11">
          <label>11</label>
          <institution>Logilab</institution>
          ,
          <country country="FR">France</country>
        </aff>
        <aff id="aff12">
          <label>12</label>
          <institution>Pistoia Alliance Inc.</institution>
          ,
          <country country="US">USA</country>
        </aff>
        <aff id="aff13">
          <label>13</label>
          <institution>TasLab, Trentino Digitale SpA</institution>
          ,
          <addr-line>Trento</addr-line>
          ,
          <country country="IT">Italy</country>
        </aff>
        <aff id="aff14">
          <label>14</label>
          <institution>University of Economics</institution>
          ,
          <addr-line>Prague</addr-line>
          ,
          <country country="CZ">Czech Republic</country>
        </aff>
        <aff id="aff15">
          <label>15</label>
          <institution>University of Mannheim</institution>
          ,
          <country country="DE">Germany</country>
        </aff>
      </contrib-group>
      <pub-date>
        <year>2020</year>
      </pub-date>
      <volume>8797</volume>
      <fpage>3416</fpage>
      <lpage>3421</lpage>
    </article-meta>
  </front>
  <body>
    <sec id="sec-1">
      <title>-</title>
      <p>Abstract. The Ontology Alignment Evaluation Initiative (OAEI) aims at
comparing ontology matching systems on precisely defined test cases. These test
cases can be based on ontologies of different levels of complexity and use
different evaluation modalities (e.g., blind evaluation, open evaluation, or consensus).
The OAEI 2020 campaign offered 12 tracks with 36 test cases, and was attended
by 19 participants. This paper is an overall presentation of that campaign.</p>
    </sec>
    <sec id="sec-2">
      <title>1 Introduction</title>
      <p>The Ontology Alignment Evaluation Initiative1 (OAEI) is a coordinated international
initiative, which organizes the evaluation of an increasing number of ontology matching
systems [26, 28], and which has been run for seventeen years by now. The main goal of
the OAEI is to compare systems and algorithms openly and on the same basis, in order
to allow anyone to draw conclusions about the best matching strategies. Furthermore,
the ambition is that, from such evaluations, developers can improve their systems and
offer better tools that answer the evolving application needs.</p>
      <p>Two first events were organized in 2004: (i) the Information Interpretation and
Integration Conference (I3CON) held at the NIST Performance Metrics for Intelligent
Systems (PerMIS) workshop and (ii) the Ontology Alignment Contest held at the
Evaluation of Ontology-based Tools (EON) workshop of the annual International Semantic
Web Conference (ISWC) [66]. Then, a unique OAEI campaign occurred in 2005 at the
workshop on Integrating Ontologies held in conjunction with the International
Conference on Knowledge Capture (K-Cap) [7]. From 2006 until the present, the OAEI
campaigns were held at the Ontology Matching workshop, collocated with ISWC [5, 4,
1, 2, 11, 18, 15, 3, 24, 23, 22, 10, 25, 27], which this year took place virtually (originally
planned in Athens, Greece)2.</p>
      <p>Since 2011, we have been using an environment for automatically processing
evaluations (Section 2.1) which was developed within the SEALS (Semantic Evaluation At
Large Scale) project3. SEALS provided a software infrastructure for automatically
executing evaluations and evaluation campaigns for typical semantic web tools, including
ontology matching. Since OAEI 2017, a novel evaluation environment, called
HOBBIT (Section 2.1), was adopted for the HOBBIT Link Discovery track, and later
extended to enable the evaluation of other tracks. Some tracks are run exclusively through
SEALS and others through HOBBIT, but several allow participants to choose the
platform they prefer. This year, the MELT framework [36] was adopted in order to facilitate
the SEALS and HOBBIT wrapping and evaluation.</p>
      <p>This paper synthesizes the 2020 evaluation campaign and introduces the results
provided in the papers of the participants. The remainder of the paper is organized as
follows: in Section 2, we present the overall evaluation methodology; in Section 3 we
present the tracks and datasets; in Section 4 we present and discuss the results; and
finally, Section 5 discusses the lessons learned.
? Copyright © 2020 for this paper by its authors. Use permitted under Creative Commons
License Attribution 4.0 International (CC BY 4.0).
1 http://oaei.ontologymatching.org
2 http://om2020.ontologymatching.org
3 http://www.seals-project.eu</p>
    </sec>
    <sec id="sec-3">
      <title>Methodology</title>
      <sec id="sec-3-1">
        <title>Evaluation platforms</title>
        <p>The OAEI evaluation was carried out in one of two alternative platforms: the SEALS
client or the HOBBIT platform. Both have the goal of ensuring reproducibility and
comparability of the results across matching systems.</p>
        <p>The SEALS client was developed in 2011. It is a Java-based command line
interface for ontology matching evaluation, which requires system developers to implement
a simple interface and to wrap their tools in a predefined way including all required
libraries and resources. A tutorial for tool wrapping is provided to the participants,
describing how to wrap a tool and how to run a full evaluation locally.</p>
        <p>The HOBBIT platform4 was introduced in 2017. It is a web interface for linked
data and ontology matching evaluation, which requires systems to be wrapped inside
docker containers and includes a SystemAdapter class, then being uploaded into the
HOBBIT platform [44].</p>
        <p>Both platforms compute the standard evaluation metrics against the reference
alignments: precision, recall and F-measure. In test cases where different evaluation
modalities are required, evaluation was carried out a posteriori, using the alignments produced
by the matching systems.</p>
        <p>The MELT framework5 [36] was introduced in 2019 and is under active
development. It allows to develop, evaluate, and package matching systems for arbitrary
evaluation interfaces like SEALS or HOBBIT. It further enables developers to use Python
in their matching systems. In terms of evaluation, MELT offers a correspondence level
analysis for multiple matching systems which can even implement different interfaces.
It is, therefore, suitable for track organisers as well as system developers.
2.2</p>
      </sec>
      <sec id="sec-3-2">
        <title>OAEI campaign phases</title>
        <p>As in previous years, the OAEI 2020 campaign was divided into three phases:
preparatory, execution, and evaluation.</p>
        <p>In the preparatory phase, the test cases were provided to participants in an initial
assessment period between June 15th and July 15th, 2020. The goal of this phase is to
ensure that the test cases make sense to participants, and give them the opportunity to
provide feedback to organizers on the test case as well as potentially report errors. At
the end of this phase, the final test base was frozen and released.</p>
        <p>During the ensuing execution phase, participants test and potentially develop their
matching systems to automatically match the test cases. Participants can self-evaluate
their results either by comparing their output with the reference alignments or by using
either of the evaluation platforms. They can tune their systems with respect to the
nonblind evaluation as long as they respect the rules of the OAEI. Participants were required
to register their systems and make a preliminary evaluation by July 31st. The execution
phase was terminated on October 15th, 2020, at which date participants had to submit
the (near) final versions of their systems (SEALS-wrapped and/or HOBBIT-wrapped).
4 https://project-hobbit.eu/outcomes/hobbit-platform/
5 https://github.com/dwslab/melt</p>
        <p>During the evaluation phase, systems were evaluated by all track organizers. In
case minor problems were found during the initial stages of this phase, they were
reported to the developers, who were given the opportunity to fix and resubmit their
systems. Initial results were provided directly to the participants, whereas final results for
most tracks were published on the respective OAEI web pages by October 24th, 2020.
3</p>
      </sec>
    </sec>
    <sec id="sec-4">
      <title>Tracks and test cases</title>
      <p>This year’s OAEI campaign consisted of 12 tracks gathering 36 test cases, all of which
included OWL ontologies to align.6 They can be grouped into:
– Schema matching tracks, which have as objective matching ontology classes and/or
properties.
– Instance Matching tracks, which have as objective matching ontology instances.
– Instance and Schema Matching tracks, which involve both of the above.
– Complex Matching tracks, which have as objective finding complex
correspondences between ontology entities.
– Interactive tracks, which simulate user interaction to enable the benchmarking of
interactive matching algorithms.</p>
      <sec id="sec-4-1">
        <title>The tracks are summarized in Table 1.</title>
        <p>3.1</p>
        <sec id="sec-4-1-1">
          <title>Anatomy</title>
          <p>The anatomy track comprises a single test case consisting of matching two fragments
of biomedical ontologies which describe the human anatomy7 (3304 classes) and the
anatomy of the mouse8 (2744 classes). The evaluation is based on a manually curated
reference alignment. This dataset has been used since 2007 with some improvements
over the years [20].</p>
          <p>Systems are evaluated with the standard parameters of precision, recall, F-measure.
Additionally, recall+ is computed by excluding trivial correspondences (i.e.,
correspondences that have the same normalized label). Alignments are also checked for
coherence using the Pellet reasoner. The evaluation was carried out on a server with a 6
core CPU @ 3.46 GHz with 8GB allocated RAM, using the SEALS client. For some
system requires more RAM, the evaluation was carried out on a Windows 10 (64-bit)
desktop with an Intel Core i7-6700 CPU @ 3.40GHz x 8 with 16GB RAM allocated.
However, the evaluation parameters were computed a posteriori, after removing from
the alignments produced by the systems, correspondences expressing relations other
than equivalence, as well as trivial correspondences in the oboInOwl namespace (e.g.,
oboInOwl#Synonym = oboInOwl#Synonym). The results obtained with the SEALS
client vary in some cases by 0.5% compared to the results presented below.</p>
        </sec>
      </sec>
      <sec id="sec-4-2">
        <title>6 The Biodiversity and Ecology track also included SKOS thesauri.</title>
        <p>7 www.cancer.gov/cancertopics/cancerlibrary/terminologyresources
8 http://www.informatics.jax.org/searches/AMA_form.shtml
open+blind DFER,, EITN, ,NELS,, SEALS</p>
        <p>AR, CZ, CN,</p>
        <p>RU, PT
Schema Matching
= [0 1]
=, &lt;=
=, &lt;=
=
=
=
Open evaluation is made with already published reference alignments and blind evaluation is
made by organizers, either from reference alignments unknown to the participants or manually.
The biodiversity and ecology (biodiv) track has been originally motivated by
two projects, namely GFBio9 (The German Federation for Biological Data) and
AquaDiva10, which aim at providing semantically enriched data management solutions
for data capture, annotation, indexing and search [46, 48]. This year, the third edition
of the biodiv track features the two matching tasks present in former editions, namely:
matching the Environment Ontology (ENVO) [9] to the Semantic Web for Earth and
Environment Technology Ontology (SWEET) [58], and matching the Flora
Phenotype Ontology (FLOPO) [38] to Plant Trait Ontology (PTO) [14]. In this edition, we
partnered with the D2KAB project11 (Data to Knowledge in Agronomy and
Biodiversity) which develops the AgroPortal12 vocabulary and ontology repository, to include
9 www.gfbio.org
10 www.aquadiva.uni-jena.de
11 www.d2kab.org
12 agroportal.lirmm.fr
EN
EN
EN
EN
EN
EN
EN
EN
EN
EN</p>
      </sec>
      <sec id="sec-4-3">
        <title>SEALS</title>
      </sec>
      <sec id="sec-4-4">
        <title>SEALS</title>
      </sec>
      <sec id="sec-4-5">
        <title>SEALS</title>
      </sec>
      <sec id="sec-4-6">
        <title>SEALS both</title>
      </sec>
      <sec id="sec-4-7">
        <title>HOBBIT HOBBIT SEALS</title>
      </sec>
      <sec id="sec-4-8">
        <title>SEALS</title>
      </sec>
      <sec id="sec-4-9">
        <title>SEALS</title>
        <p>SEALS
two new matching tasks involving important thesauri (originally developed in SKOS)
in agronomy and environmental sciences: finding alignments between the AGROVOC
thesaurus [59] and the US National Agricultural Library Thesaurus (NALT)13 and
between the General Multilingual Environmental Thesaurus (GEMET)14 and the Analysis
and Experimentation on Ecosystems thesaurus (ANAEETHES)[13]. These ontologies
and thesauri are particularly useful for biodiversity and ecology research and are
being used in various projects. They have been developed in parallel and are significantly
overlapping. They are semantically rich and contain tens of thousands of concepts. By
providing semantic resources developed in SKOS, our objective is also to encourage the
ontology alignment community to develop tools that can natively handle SKOS which
is an important standard to encode terminologies (particularly thesauri and taxonomies)
and for which alignment is also very important.</p>
        <p>Table 2 presents detailed information about the ontologies and thesauri used in the
evaluation, such as the ontology format, version, number of classes as well as the
number of instances15.</p>
        <p>
          For the ontologies ENVO, SWEET, FLOPO and PTO, we created the reference
alignments for the tasks following the same procedure as in former editions. Reference
files were produced using a hybrid approach consisting of (
          <xref ref-type="bibr" rid="ref1">1</xref>
          ) a consensus alignment
based on matching systems output, then (
          <xref ref-type="bibr" rid="ref2">2</xref>
          ) manually validating a subset of unique
mappings produced by each system (and adding them to the consensus if considered
correct), and finally (
          <xref ref-type="bibr" rid="ref3">3</xref>
          ) adding a set of manually generated correspondences. The matching
systems used to generate the consensus alignments were those participating to this track
in 2018 [4], namely: AML, Lily, the LogMap family, POMAP and XMAP.
13 agclass.nal.usda.gov
14 www.eionet.europa.eu/gemet
15 Note that SKOS thesauri conceptualize by means of instances of skos:Concept and not
owl:Class. Still, the biodiv track is different from instance matching tracks, as in both
cases concepts or classes are used to define the structure (or schema) of a semantic resource.
        </p>
        <p>For the thesauri AGROVOC, NALT, GEMET and ANEETHES, we created the
reference alignments using the Ontology Mapping Harvesting Tool (OMHT).16 OMHT
was developed as a standalone Java program that works with one semantic resource file
pulled out from AgroPortal or BioPortal17. OMHT automatically extracts all declared
mappings by developers inside an ontology or a thesauri source files. We used for the
reference alignments only the mappings with a skos:exactMatch property.</p>
        <p>The evaluation was carried out on a Windows 10 (64-bit) desktop with an Intel Core
i7-4770 CPU @ 3.40GHz x 4 with 16 GB RAM allocated, using the SEALS client.
Systems were evaluated using the standard metrics.
3.3</p>
        <sec id="sec-4-9-1">
          <title>Conference</title>
          <p>The conference track features a single test case that is a suite of 21 matching tasks
corresponding to the pairwise combination of 7 moderately expressive ontologies describing
the domain of organizing conferences. The dataset and its usage are described in [70].</p>
          <p>The track uses several reference alignments for evaluation: the old (and not fully
complete) manually curated open reference alignment, ra1; an extended, also
manually curated version of this alignment, ra2; a version of the latter corrected to resolve
violations of conservativity, rar2; and an uncertain version of ra1 produced through
crowd-sourcing, where the score of each correspondence is the fraction of people in
the evaluation group that agree with the correspondence. The latter reference was used
in two evaluation modalities: discrete and continuous evaluation. In the former,
correspondences in the uncertain reference alignment with a score of at least 0.5 are treated
as correct whereas those with lower score are treated as incorrect, and standard
evaluation parameters are used to evaluated systems. In the latter, weighted precision, recall
and F-measure values are computed by taking into consideration the actual scores of
the uncertain reference, as well as the scores generated by the matching system. For the
sharp reference alignments (ra1, ra2 and rar2), the evaluation is based on the standard
parameters, as well the F0:5-measure and F2-measure and on conservativity and
consistency violations. Whereas F1 is the harmonic mean of precision and recall where both
receive equal weight, F2 gives higher weight to recall than precision and F0:5 gives
higher weight to precision higher than recall. The track also includes an analysis of
False Positives.</p>
          <p>Two baseline matchers are used to benchmark the systems: edna string edit distance
matcher; and StringEquiv string equivalence matcher as in the anatomy test case.</p>
          <p>The evaluation was carried out on a Windows 10 (64-bit) desktop with an Intel
Core i7–8550U (1,8 GHz, TB 4 GHz) x 4 with 16 GB RAM allocated using the SEALS
client. Systems were evaluated using the standard metrics.
3.4</p>
        </sec>
        <sec id="sec-4-9-2">
          <title>Disease and Phenotype</title>
          <p>The Disease and Phenotype is organized by the Pistoia Alliance Ontologies Mapping
project team18. It comprises 2 test cases that involve 4 biomedical ontologies
cov16 https://github.com/agroportal/ontology_mapping_harvester
17 https://bioportal.bioontology.org
18 http://www.pistoiaalliance.org/projects/ontologies-mapping/
ering the disease and phenotype domains: Human Phenotype Ontology (HP) versus
Mammalian Phenotype Ontology (MP) and Human Disease Ontology (DOID) versus
Orphanet and Rare Diseases Ontology (ORDO). Currently, correspondences between
these ontologies are mostly curated by bioinformatics and disease experts who would
benefit from automation of their workflows supported by implementation of
ontology matching algorithms. More details about the Pistoia Alliance Ontologies Mapping
project and the OAEI evaluation are available in [31]. Table 3 summarizes the versions
of the ontologies used in OAEI 2020.</p>
          <p>The reference alignments used in this track are silver standard consensus
alignments automatically built by merging/voting the outputs of the participating systems
in the OAEI campaigns 2016-2020 (with vote=3). Note that systems participating with
different variants and in different years only contributed once in the voting, that is, the
voting was done by family of systems/variants rather than by individual systems. The
HP-MP silver standard thus contains 2,504 correspondences, whereas the DOID-ORDO
one contains 3,909 correspondences.</p>
          <p>Systems were evaluated using the standard parameters as well as the (approximate)
number of unsatisfiable classes computed using the OWL 2 EL reasoner ELK [47]. The
evaluation was carried out in a Ubuntu 18 Laptop with an Intel Core i5-6300HQ CPU
@ 2.30GHz x 4 and allocating 15 Gb of RAM.
3.5</p>
        </sec>
        <sec id="sec-4-9-3">
          <title>Large Biomedical Ontologies</title>
          <p>The large biomedical ontologies (largebio) track aims at finding alignments between
the large and semantically rich biomedical ontologies FMA, SNOMED-CT, and NCI,
which contain 78,989, 306,591 and 66,724 classes, respectively. The track consists of
six test cases corresponding to three matching problems (FMA-NCI, FMA-SNOMED
and SNOMED-NCI) in two modalities: small overlapping fragments and whole
ontologies (FMA and NCI) or large fragments (SNOMED-CT).</p>
          <p>The reference alignments used in this track are derived directly from the UMLS
Metathesaurus [8] as detailed in [42], then automatically repaired to ensure logical
coherence. However, rather than use a standard repair procedure of removing
problem causing correspondences, we set the relation of such correspondences to “?”
(unknown). These “?” correspondences are neither considered positive nor negative when
evaluating matching systems, but are simply ignored. This way, systems that do not
perform alignment repair are not penalized for finding correspondences that (despite
causing incoherences) may or may not be correct, and systems that do perform alignment
repair are not penalized for removing such correspondences. To avoid any bias,
correspondences were considered problem causing if they were selected for removal by any
of the three established repair algorithms: Alcomo [52], LogMap [41], or AML [60].
The reference alignments are summarized in Table 4.</p>
          <p>The evaluation was carried out in a Ubuntu 18 Laptop with an Intel Core i5-6300HQ
CPU @ 2.30GHz x 4 and allocating 15 Gb of RAM. Evaluation was based on the
standard parameters (modified to account for the “?” relations) as well as the number
of unsatisfiable classes and the ratio of unsatisfiable classes with respect to the size of
the union of the input ontologies. Unsatisfiable classes were computed using the OWL
2 reasoner HermiT [54], or, in the cases in which HermiT could not cope with the
input ontologies and the alignments (in less than 2 hours) a lower bound on the number
of unsatisfiable classes (indicated by ) was computed using the OWL2 EL reasoner
ELK [47].
3.6</p>
        </sec>
        <sec id="sec-4-9-4">
          <title>Multifarm</title>
          <p>The multifarm track [53] aims at evaluating the ability of matching systems to deal with
ontologies in different natural languages. This dataset results from the translation of 7
ontologies from the conference track (cmt, conference, confOf, iasted, sigkdd, ekaw and
edas) into 10 languages: Arabic (ar), Chinese (cn), Czech (cz), Dutch (nl), French (fr),
German (de), Italian (it), Portuguese (pt), Russian (ru), and Spanish (es). The dataset
is composed of 55 pairs of languages, with 49 matching tasks for each of them, taking
into account the alignment direction (e.g. cmten !edasde and cmtde !edasen are
distinct matching tasks). While part of the dataset is openly available, all matching tasks
involving the edas and ekaw ontologies (resulting in 55 24 matching tasks) are used
for blind evaluation.</p>
          <p>We consider two test cases: i) those tasks where two different ontologies
(cmt!edas, for instance) have been translated into two different languages; and ii)
those tasks where the same ontology (cmt!cmt) has been translated into two
different languages. For the tasks of type ii), good results are not only related to the use of
specific techniques for dealing with cross-lingual ontologies, but also on the ability to
exploit the identical structure of the ontologies.</p>
          <p>The reference alignments used in this track derive directly from the manually
curated Conference ra1 reference alignments. The systems have been executed on a
Ubuntu Linux machine configured with 8GB of RAM running under a Intel Core CPU
2.00GHz x4 processors, using the SEALS client.
3.7
The Link Discovery track features two test cases, Linking and Spatial, that deal with
link discovery for spatial data represented as trajectories i.e., sequences of
longitude, latitude pairs. The track is based on two datasets generated from TomTom19 and
Spaten [17].</p>
          <p>The Linking test case aims at testing the performance of instance matching tools
that implement mostly string-based approaches for identifying matching entities. It
can be used not only by instance matching tools, but also by SPARQL engines that
deal with query answering over geospatial data. The test case was based on
SPIMBENCH [62], but since the ontologies used to represent trajectories are fairly simple
and do not consider complex RDF or OWL schema constructs already supported by
SPIMBENCH, only a subset of the transformations implemented by SPIMBENCH was
used. The transformations implemented in the test case were (i) string-based with
different (a) levels, (b) types of spatial object representations and (c) types of date
representations, and (ii) schema-based, i.e., addition and deletion of ontology (schema) properties.
These transformations were implemented in the TomTom dataset. In a nutshell, instance
matching systems are expected to determine whether two traces with their points
annotated with place names designate the same trajectory. In order to evaluate the systems
a ground truth was built that contains the set of expected links where an instance s1 in
the source dataset is associated with an instance t1 in the target dataset that has been
generated as a modified description of s1.</p>
          <p>The Spatial test case aims at testing the performance of systems that deal with
topological relations proposed in the state of the art DE-9IM (Dimensionally Extended
nine-Intersection Model) model [65]. The benchmark generator behind this test case
implements all topological relations of DE-9IM between trajectories in the two
dimensional space. To the best of our knowledge such a generic benchmark, that takes as
input trajectories and checks the performance of linking systems for spatial data does
not exist. The focus for the design was (a) on the correct implementation of all the
topological relations of the DE-9IM topological model and (b) on producing datasets large
enough to stress the systems under test. The supported relations are: Equals, Disjoint,
Touches, Contains/Within, Covers/CoveredBy, Intersects, Crosses, Overlaps. The test
case comprises tasks for all the DE-9IM relations and for LineString/LineString and
LineString/Polygon cases, for both TomTom and Spaten datasets, ranging from 200 to
2K instances. We did not exceed 64 KB per instance due to a limitation of the Silk
system20, in order to enable a fair comparison of the systems participating in this track.</p>
          <p>The evaluation for both test cases was carried out using the HOBBIT platform.
3.8</p>
        </sec>
        <sec id="sec-4-9-5">
          <title>SPIMBENCH</title>
          <p>The SPIMBENCH track consists of matching instances that are found to refer to the
same real-world entity corresponding to a creative work (that can be a news item,
19 https://www.tomtom.com/en_gr/
20 https://github.com/silk-framework/silk/issues/57
blog post or programme). The datasets were generated and transformed using
SPIMBENCH [62] by altering a set of original linked data through value-based,
structurebased, and semantics-aware transformations (simple combination of transformations).
They share almost the same ontology (with some differences in property level, due
to the structure-based transformations), which describes instances using 22 classes, 31
data properties, and 85 object properties. Participants are requested to produce a set of
correspondences between the pairs of matching instances from the source and target
datasets that are found to refer to the same real-world entity. An instance in the source
dataset can have none or one matching counterpart in the target dataset. The
SPIMBENCH task uses two sets of datasets21 with different scales (i.e., number of instances
to match):
– Sandbox (380 INSTANCES, 10000 TRIPLES). It contains two datasets called
source (Tbox1) and target (Tbox2) as well as the set of expected correspondences
(i.e., reference alignment).
– Mainbox (1800 CWs, 50000 TRIPLES). It contains two datasets called source
(Tbox1) and target (Tbox2). This test case is blind, meaning that the reference
alignment is not given to the participants.</p>
          <p>In both cases, the goal is to discover the correspondences among the instances in the
source dataset (Tbox1) and the instances in the target dataset (Tbox2).</p>
          <p>The evaluation was carried out using the HOBBIT platform.
3.9</p>
        </sec>
        <sec id="sec-4-9-6">
          <title>Geolink Cruise</title>
          <p>The Geolink Cruise track consists of matching instances from different ontologies
describing the same cruise in the real-world. The datasets are collected from the Geolink
project,22 which was funded under the U.S. National Science Foundation’s EarthCube
initiative. The datasets and alignments are guaranteed to contain real-world use cases to
solve the instance matching problem in practice. In the GeoLink Cruise dataset, there
are two ontologies which are GeoLink Base Ontology (gbo) and GeoLink Modular
Ontology (gmo). The data providers from different organizations populate their own
data into these two ontologies. In this track, we utilize instances from two different
data providers, Biological and Chemical Oceanography Data Management Office
(bco21 Although the files are called Tbox1 and Tbox2, they actually contain a Tbox and an Abox.
22 https://www.geolink.org/
dmo)23 and Rolling Deck to Repository (r2r)24 and populate all the triples related to
Cruise into two ontologies. There are 491 Cruise pairs between these two datasets that
are labelled by domain experts as equivalent. Some statistic information of the
ontologies are listed in the Table 5. More details of this benchmark can be found in the paper
[6].
The Knowledge Graph track was run for the third year. The task of the track is to match
pairs of knowledge graphs, whose schema and instances have to be matched
simultaneously. The individual knowledge graphs are created by running the DBpedia extraction
framework on eight different Wikis from the Fandom Wiki hosting platform25 in the
course of the DBkWik project [34, 33]. They cover different topics (movies, games,
comics and books) and three Knowledge Graph clusters sharing the same domain e.g.
star trek, as shown in Table 6.</p>
          <p>The evaluation is based on reference correspondences at both schema and instance
levels. While the schema level correspondences were created by experts, the instance
correspondences were extracted from the wiki page itself. Due to the fact that not all
inter wiki links on a page represent the same concept a few restrictions were made: 1)
only links in sections with a header containing “link” are used, 2) all links are removed
where the source page links to more than one concept in another wiki (ensures the
alignments are functional), 3) multiple links which point to the same concept are also
removed (ensures injectivity), 4) links to disambiguation pages were manually checked
and corrected. Since we do not have a correspondence for each instance, class, and
property in the graphs, this gold standard is only a partial gold standard.</p>
          <p>The evaluation was executed on a virtual machine (VM) with 32GB of RAM and
16 vCPUs (2.4 GHz), with Debian 9 operating system and Openjdk version 1.8.0 265,
23 https://www.bco-dmo.org/
24 https://www.rvdata.us/
25 https://www.wikia.com/
using the SEALS client (version 7.0.5). The -o option in SEALS is used to provide
the two knowledge graphs which should be matched. This decreases runtime because
the matching system can load the input from local files rather than downloading it from
HTTP URLs. We could not use the ”-x” option of SEALS because the evaluation routine
needed to be changed for two reasons: first, to differentiate between results for class,
property, and instance correspondences, and second, to deal with the partial nature of
the gold standard.</p>
          <p>The alignments were evaluated based on precision, recall, and f-measure for classes,
properties, and instances (each in isolation). The partial gold standard contained 1:1
correspondences and we further assume that in each knowledge graph, only one
representation of the concept exists. This means that if we have a correspondence in our
gold standard, we count a correspondence to a different concept as a false positive. The
count of false negatives is only increased if we have a 1:1 correspondence and it is not
found by a matcher. The whole source code for generating the evaluation results is also
available.26</p>
          <p>Additionally we run the matchers on three hidden test cases where the source wikis
are: Marvel Cinematic Universe, Memory Alpha, and Star Wars Wiki. The target wiki is
for all test cases the same. It is the lyrics wiki with 1,062,920 instances, 270 properties
and 67 classes. The goal is to explore how the matchers behave on matching mostly
unrelated knowledge graphs.</p>
          <p>As a baseline, we employed two simple string matching approaches. The source
code for these matchers is publicly available.27</p>
        </sec>
        <sec id="sec-4-9-7">
          <title>3.11 Interactive Matching</title>
          <p>The interactive matching track aims to assess the performance of semi-automated
matching systems by simulating user interaction [56, 19, 50]. The evaluation thus
focuses on how interaction with the user improves the matching results. Currently, this
track does not evaluate the user experience or the user interfaces of the systems [39,
19].</p>
          <p>The interactive matching track is based on the datasets from the Anatomy and
Conference tracks, which have been previously described. It relies on the SEALS client’s
Oracle class to simulate user interactions. An interactive matching system can present
a collection of correspondences simultaneously to the oracle, which will tell the system
whether that correspondence is correct or not. If a system presents up to three
correspondences together and each correspondence presented has a mapped entity (i.e., class
or property) in common with at least one other correspondence presented, the oracle
counts this as a single interaction, under the rationale that this corresponds to a
scenario where a user is asked to choose between conflicting candidate correspondences.
To simulate the possibility of user errors, the oracle can be set to reply with a given
error probability (randomly, from a uniform distribution). We evaluated systems with
four different error rates: 0.0 (perfect user), 0.1, 0.2, and 0.3.
26 http://oaei.ontologymatching.org/2020/results/knowledgegraph/
matching-eval-trackspecific.zip
27 http://oaei.ontologymatching.org/2019/results/knowledgegraph/
kgBaselineMatchers.zip</p>
          <p>In addition to the standard evaluation parameters, we also compute the number of
requests made by the system, the total number of distinct correspondences asked, the
number of positive and negative answers from the oracle, the performance of the system
according to the oracle (to assess the impact of the oracle errors on the system) and
finally, the performance of the oracle itself (to assess how erroneous it was).</p>
          <p>The evaluation was carried out on a server with 3.46 GHz (6 cores) and 8GB RAM
allocated to the matching systems. For systems requiring more RAM, the evaluation
was carried out on a Windows 10 (64-bit) desktop with an Intel Core i7-6700 CPU
@ 3.40GHz x 8 with 16GB RAM allocated. Each system was run ten times and the
final result of a system for each error rate represents the average of these runs. For
the Conference dataset with the ra1 alignment, precision and recall correspond to the
micro-average over all ontology pairs, whereas the number of interactions is the total
number of interactions for all the pairs.
3.12</p>
        </sec>
        <sec id="sec-4-9-8">
          <title>Complex Matching</title>
          <p>The complex matching track is meant to evaluate the matchers based on their
ability to generate complex alignments. A complex alignment is composed of
complex correspondences typically involving more than two ontology entities, such as
o1:AcceptedPaper o2:Paper u o2:hasDecision.o2:Acceptance. In addition to last
year’s datasets [69], two new datasets have been added: Populated Geolink and
Populated Enslaved.</p>
          <p>The complex conference dataset is composed of three ontologies: cmt, conference
and ekaw from the conference dataset. The reference alignment was created as a
consensus between experts. In the evaluation process, the matchers can take the simple
reference alignment ra1 as input. The precision and recall measures are manually
calculated over the complex equivalence correspondences only.</p>
          <p>The populated complex conference is a populated version of the Conference
dataset. 5 ontologies have been populated with more or less common instances
resulting in 6 datasets (6 versions on the seals repository: v0, v20, v40, v60, v80 and v100).
The alignments were evaluated based on Competency Questions for Alignment, i.e.,
basic queries that the alignment should be able to cover [67]. The queries are
automatically rewritten using 2 systems: that from [68] which covers (1:n) correspondences with
EDOAL expressions; and a system which compares the answers (sets of instances or
sets of pairs of instances) of the source query and the source member of the
correspondences and which outputs the target member if both sets are identical. The best rewritten
query scores are kept. A precision score is given by comparing the instances described
by the source and target members of the correspondences.</p>
          <p>The Hydrography dataset consists of matching four different source ontologies
(hydro3, hydrOntology-translated, hydrOntology-native, and cree) to a single target
ontology (SWO) [12]. The evaluation process is based on three subtasks: given an entity
from the source ontology, identify all related entities in the source and target ontology;
given an entity in the source ontology and the set of related entities, identify the logical
relation that holds between them; identify the full complex correspondences. The three
subtasks were evaluated based on relaxed precision and recall [21].</p>
          <p>The GeoLink dataset derives from the homonymous project, funded under the U.S.
National Science Foundation’s EarthCube initiative. It is composed of two ontologies:
the GeoLink Base Ontology (GBO) and the GeoLink Modular Ontology (GMO). The
GeoLink project is a real-world use case of ontologies. The alignment between the two
ontologies was developed in consultation with domain experts from several geoscience
research institutions. More detailed information on this benchmark can be found in [72].
Evaluation was done in the same way as with the Hydrography dataset. The evaluation
platform was a MacBook Pro with a 2.5 GHz Intel Core i7 processor and 16 GB of
1600 MHz DDR3 RAM running mac OS Catalina version 10.15.6.</p>
          <p>The Populated GeoLink dataset is designed to allow alignment systems that rely on
the instance data to participate over the Geolink benchmark. The instance data are from
real-worlds and collected from seven data repositories in the Geolink project. More
detailed information on this benchmark can be found in [73]. Evaluation was done in the
same way as with the Hydrography dataset. The evaluation platform was a MacBook
Pro with a 2.5 GHz Intel Core i7 processor and 16 GB of 1600 MHz DDR3 RAM
running mac OS Catalina version 10.15.6.</p>
          <p>The Populated Enslaved dataset was derived from the ongoing project entitled
“Enslaved: People of the Historical Slave Trade28 and funded by The Andrew W.
Mellon Foundation where the focus is on tracking the movements and details of peoples
in the historical slave trade. It is composed of the Enslaved ontology and the Enslaved
Wikibase repository along with the populated instance data. To the best of our
knowledge, it is the first attempt to align a modular ontology to the Wikibase repository. More
detailed information on this benchmark can be found in [71]. Evaluation was done in
the same way as with the Hydrography dataset. The evaluation platform was a
MacBook Pro with a 2.5 GHz Intel Core i7 processor and 16 GB of 1600 MHz DDR3 RAM
running mac OS Catalina version 10.15.6.</p>
          <p>The Taxon dataset is composed of four knowledge bases containing knowledge
about plant taxonomy: AgronomicTaxon, AGROVOC, TAXREF-LD and DBpedia. The
evaluation is two-fold: first, the precision of the output alignment is manually assessed;
then, a set of source queries are rewritten using the output alignment. The rewritten
target query is then manually classified as correct or incorrect. A source query is
considered successfully rewritten if at least one of the target queries is semantically equivalent
to it. The proportion of source queries successfully rewritten is then calculated (QWR
in the results table). The evaluation over this dataset is open to all matching systems
(simple or complex) but some queries can not be rewritten without complex
correspondences. The evaluation was performed with an Ubuntu 16.04 machine configured with
16GB of RAM running under a i7-4790K CPU 4.00GHz x 8 processors.
4
4.1</p>
        </sec>
      </sec>
    </sec>
    <sec id="sec-5">
      <title>Results and Discussion</title>
      <sec id="sec-5-1">
        <title>Participation</title>
        <p>Following an initial period of growth, the number of OAEI participants has remained
approximately constant since 2012, which is slightly over 20. This year we count with
28 https://enslaved.org/
19 participating systems. Table 7 lists the participants and the tracks in which they
competed. Some matching systems participated with different variants (AML, LogMap)
whereas others were evaluated with different configurations, as requested by developers
(see test case sections for details).
tseySm ILNA ec2LODAV LAM LCAM ROAA oxTBA SEKDM RCNAA IFLTRM ilyL agopLM i-agopoLBM tagopLLM ttcenoonnCO RADON i-renERm ilkS lieegnVA ttrckhMW tlao19=T
Confidence X X X X X X X X X X X X X X X X 16
anatomy # # # # # # # # 11
conference # # # # # # # # # 10
multifarm # # # # # # # # #G # # # # # 6</p>
        <p>complex # # # # # # # # # # # # # # # # 3
interactive # # # # # # # # # # # # # # # # 3</p>
        <p>largebio # G# # # #G # # # # # # # # 8
phenotype # # # # # # # # # # # # 7</p>
        <p>biodiv # G# # # # # # # # # # # # #G 7
spimbench # # # # # # # # # # # # # # 5
link discovery # # # # # # # # # # # # # # # # 3
geolink cruise # # # # # # # # # # # # # # # # # # # 0
knowledge graph # # # # # # #G # # # # # 8
total 3 6 10 1 1 6 4 1 1 4 9 5 7 1 1 1 1 2 7 71
Confidence pertains to the confidence scores returned by the system, with X indicating that they
are non-boolean; # indicates that the system did not participate in the track; indicates that it
participated fully in the track; and G# indicates that it participated in or completed only part of the
tasks of the track.</p>
        <p>A number of participating systems use external sources of background knowledge,
which are especially critical in matching ontologies in the biomedical domain.
LogMapBio uses BioPortal as mediating ontology provider, that is, it retrieves from BioPortal
the most suitable top-10 ontologies for each matching task. LogMap uses
normalizations and spelling variants from the general (biomedical) purpose SPECIALIST
Lexicon. AML has three sources of background knowledge which can be used as mediators
between the input ontologies: the Uber Anatomy Ontology (Uberon), the Human
Disease Ontology (DOID) and the Medical Subject Headings (MeSH). XMAP and Lily
use a dictionary of synonyms (pre)extracted from the UMLS Metathesaurus. In
addition Lily also uses a dictionary of synonyms (pre)extracted from BioPortal.
4.2
The results for the Anatomy track are shown in Table 8. Of the 11 systems participating
in the Anatomy track, 10 achieved an F-measure higher than the StringEquiv baseline.
Three systems were first time participants (ATBox, OntoConnect, and DESKMatcher).
Long-term participating systems showed few changes in comparison with previous
years with respect to alignment quality (precision, recall, F-measure, and recall+), size
and run time. The exceptions were ALIN which increased in precision (from 0.974
to 0.986), recall (from 0.698 to 0.72), recall+ (from 0.365 to 0.382), F-measure (from
0.813 to 0.832), and size (from 1086 to 1107), and Lily that increased in precision (from
0.873 to 0.901), recall (from 0.796 to 0.902), recall+ (from 0.52 to 0.747), F-measure
(from 0.833 to 0.901), and size (from 1381 to 1517). In terms of run time, 4 out of 11
systems computed an alignment in less than 100 seconds, a ratio which is similar to
2019 (5 out of 12). LogMapLite remains the system with the shortest runtime.
Regarding quality, AML remains the system with the highest F-measure (0.941) and recall+
(0.81), but 3 other systems obtained an F-measure above 0.88 (Lily, LogMapBio, and
LogMap) which is at least as good as the best systems in OAEI 2007-2010. Like in
previous years, there is no significant correlation between the quality of the generated
alignment and the run time. Four systems produced coherent alignments.
Four systems participating this year did participate to this track last year as well: AML
and the LogMap family systems (LogMap, LogMapBio and LogMapLT). Three are
new participants: ATBox, ALOD2Vec and Wiktionary. The newcomer ATBox did not
register explicitly to the track but could cope with at least one task so we did include
its results. As in the previous edition, we used precision, recall and F-measure to
evaluate the performance of the participating systems. The results for the Biodiversity and
Ecology track are shown in Table 9.</p>
        <p>In comparison to previous years, we observed a decrease in the number of systems
that succeeded to generate alignments for the ENVO-SWEET and FLOPO-PTO tasks.
Basically, except of AML and the LogMap variants, only ATBox could cope with the
LogMap
LogMapBio
AML
LogMapLt
ATBox
Wiktionary
ALOD2Vec
AML
LogMapLt
ATBox
LogMap
LogMapBio</p>
        <sec id="sec-5-1-1">
          <title>LogMapBio LogMap AML LogMapLt</title>
          <p>AML
tasks with fair results. ALOD2Vec and Wiktionary generated a similar, huge set of non
meaningful mappings with a very low F-measure as shown in Table 9.</p>
          <p>The results of the participating systems have slightly increased in terms of
Fmeasure for both first two tasks compared to last year. In terms of run time, Wiktionary,
ALOD2Vec and LogMapBio took the longer time, for the latter due to the loading of
mediating ontologies from BioPortal.</p>
          <p>For the FLOPO-PTO task, LogMap and LogMapBio achieved the highest
Fmeasure. AML generated a large number of mappings (significantly bigger than the
size of the reference alignment), those alignments were mostly subsumption ones. In
order to evaluate the precision in a more significant manner, we had to calculate an
approximation by manually assessing a subset of around 100 mappings, that were not
present in the reference alignment. LogMapLt and ATBox achieved a high precision
but the lowest recall.</p>
          <p>Regarding the ENVO-SWEET task, AML ranked first in terms of F-measure,
followed by LogMapLt and ATBox. The systems with the highest precision (LogMap and
LogMapBio) achieve the lowest recall. Again here, AML generated a bigger set with
a high number of subsumption mappings, it still achieved the best F-Measure for the
task. It is worth nothing that due the specific structure of the SWEET ontology, a lot of
the false positives come from homonyms [45].</p>
          <p>The ANAEETHES-GEMET and AGROVOC-NALT matching tasks have been
introduced to the track this year, with the particularity of being resources developed in
SKOS. Only AML could handle the files in their original format. LogMap and its
variants could generate mappings for ANAEETHES-GEMET, based on ontology files after
being transformed automatically into OWL. For the transformation, we made use of
a source code29 that was directly derived from AML ontology parsing module, kindly
provided to us by its developers. LogMap and LogMapBio achieve the best results with
LogMap processing the task in a shorter time. LogMapBio took a much longer time due
to downloading 10 mediating ontologies from BioPortal, still the gain is not significant
in terms of performance. The AGROVOC-NALT task has been managed only by AML.
All other systems failed in generating mappings on both the SKOS and OWL versions
of the thesauri. AML achieves good results and a very high precicion. It generated a
higher number of mappings (around 1000 more) than the curated reference alignment.
We performed a manual assessment of a subset of those mappings to reevaluate the
precision and F-measure.</p>
          <p>Overall, in this third evaluation, the results obtained from participating systems for
the two tasks ENVO-SWEET and FLOPO-PTO remained similar with a slight increase
in terms of F-measure compared to last year. The results of the two new tracks
demonstrate systems (beside AML) are not ready to handle SKOS. Sometimes automatically
transforming to OWL helps to avoid the issue, sometimes not. The number of mappings
in the AGROVOC-NALT track is really a challenge and AML does not loose in
performance which demonstrates that besides being the more tolerant tool in terms of format,
it also scales up to large size thesauri.
The conference evaluation results using the sharp reference alignment rar2 are shown
in Table 10. For the sake of brevity, only results with this reference alignment and
considering both classes and properties are shown. For more detailed evaluation results,
please check conference track’s web page.</p>
          <p>With regard to two baselines we can group tools according to system’s position:
eight matching systems outperformed both baselines (ALIN, AML, ALOD2Vec,
ATBox, LogMap, LogMapLt, VeeAlign and Wiktionary); two performed worse than both
baselines (DESKMatcher and Lily). Three matchers (ALIN and Lily) do not match
properties at all. Naturally, this has a negative effect on their overall performance.</p>
          <p>The performance of all matching systems regarding their precision, recall and
F1measure is plotted in Figure 1. Systems are represented as squares or triangles, whereas
the baselines are represented as circles.</p>
          <p>With respect to logical coherence [63, 64], as the last year, only three tools (ALIN,
AML and LogMap) have no consistency principle violation.</p>
          <p>As the last year we performed analysis of the False Positives, i.e. correspondences
discovered by the tools which were evaluated as incorrect. The list of the False Positives
29 http://oaei.ontologymatching.org/2020/biodiv/code/SKOS2OWL.zip
is available on the conference track’s web page as well as further details about this
evaluation. Comparing to the previous year we added the comparison of ”why was an
alignment discovered” assigned by us with the explanation for the alignment provided
by the system itself. This year three systems generated explanations with the mappings
ALOD2Vec, DESKMatcher and Wiktionary.</p>
          <p>The Conference evaluation results using the uncertain reference alignments are
presented in Table 11. Out of the 10 alignment systems, three (ALIN, DESKMatcher,
LogMapLt) use 1.0 as the confidence value for all matches they identify. The remaining
7 systems (ALOD2Vec, AML, ATBOX, Lily, LogMap, VeeAlign, Wiktionary) have a
wide variation of confidence values.</p>
          <p>The Multifarm evaluation results based on the blind dataset are presented in
Table 15. They have been computed using the Alignment API 4.9 and can slightly differ
from those computed with the SEALS client. We haven’t applied any threshold on the
results. We do not report the results of non-specific systems here, as we could observe
in the last campaigns that they can have intermediate results in the “same ontologies”
task (ii) and poor performance in the “different ontologies” task (i). The detailed results
can be investigated on the page of multifarm track results32.</p>
          <p>Time #pairs SizTeypePr(ei)c.– 22 Fte-smts. per pRaeicr.</p>
          <p>AML outperforms all other systems in terms of F-measure for task i) (same
behaviour in the last campaigns). In terms of precision, Wiktionary is the system that
generates the most precise alignments, followed by LogMap, VeeAlign and AML. With
respect to the task ii) LogMap has the overall best performance. Comparing the results
from last year, in terms F-measure (cases of type i), AML maintains its overall
performance (.45 in 2019, .46 in 2018, .46 in 2017, .45 in 2016 and .47 in 2015). The same
could be observed for LogMap (.37 in 2019, .37 in 2018, .36 in 2017, and .37 in 2016).
The performance in terms of F-measure of Wiktionary also remains stable. In terms
of runtime, the results are not really comparable with the ones in the last campaign
considering the fact the SEALS repositories have been moved to another server with a
different configuration.</p>
          <p>Overall, the F-measure for blind tests remains relatively stable across campaigns. As
observed in previous campaigns, systems still privilege precision over recall.
Furthermore, the overall results in MultiFarm are lower than the ones obtained for the original
English version of the Conference dataset.
32 http://oaei.ontologymatching.org/2020/results/multifarm/index.</p>
          <p>html
This year the Link Discovery track counted three participants in the Spatial test case:
AML, Silk and RADON. Those were the exact same systems (and versions) that
participated on OAEI 2019.</p>
          <p>We divided the Spatial test cases into four suites. In the first two suites (SLL and
LLL), the systems were asked to match LineStrings to LineStrings considering a given
relation for 200 and 2K instances for the TomTom and Spaten datasets. In the last two
tasks (SLP, LLP), the systems were asked to match LineStrings to Polygons (or
Polygons to LineStrings depending on the relation) again for both datasets. Since the
precision, recall and F-measure results from all systems were equal to 1.0, we are only
presenting results regarding the time performance. The time performance of the
matching systems in the SLL, LLL, SLP and LLP suites are shown in Figures 2-3. The results
can also be found in HOBBIT git (https://hobbit-project.github.io/
OAEI_2020.html).</p>
          <p>In the SLL suite, RADON has the best performance in most cases except for the
Touches and Intersects relations, followed by AML. Silk seems to need the most time,
particularly for Touches and Intersects relations in the TomTom dataset and Overlaps
in both datasets.</p>
          <p>In the LLL suite we have a more clear view of the capabilities of the systems with
the increase in the number of instances. In this case, RADON and Silk have similar
behavior as in the small dataset, but it is more clear that the systems need much more time
to match instances from the TomTom dataset. RADON has still the best performance in
most cases. AML has the next best performance and is able to handle some cases better
than other systems (e.g. Touches and Intersects), however, it also hits the platform time
limit in the case of Disjoint.</p>
          <p>In the SLP suite, in contrast to the first two suites, RADON has the best performance
for all relations. AML and Silk have minor time differences and, depending on the case,
one is slightly better than the other. All the systems need more time for the TomTom
dataset but due to the small size of the instances the time difference is minor.</p>
          <p>In the LLP suite, RADON again has the best performance in all cases. AML hits the
platform time limit in Disjoint relations on both datasets and is better than Silk in most
cases except Contains and Within on the TomTom dataset where it needs an excessive
amount of time.</p>
          <p>Taking into account the executed test cases we can identify the capabilities of the
tested systems as well as suggest some improvements. All the systems participated in
most of the test cases, with the exception of Silk which did not participate in the Covers
and Covered By test cases.</p>
          <p>RADON was the only system that successfully addressed all the tasks, and had the
best performance for the SLP and LLP suites, but it can be improved for the Touches
and Intersects relations for the SLL and LLL suites. AML performs extremely well in
most cases, but can be improved in the cases of Covers/Covered By and Contains/Within
when it comes to LineStrings/Polygons Tasks and especially in Disjoint relations where
it hits the platform time limit. Silk can be improved for the Touches, Intersects and
Overlaps relations and for the SLL and LLL tasks and for the Disjoint relation in SLP
and LLP Tasks.</p>
          <p>In general, all systems needed more time to match the TomTom dataset than the
Spaten one, due to the smaller number of points per instance in the latter. Comparing the
LineString/LineString to the LineString/Polygon Tasks we can say that all the systems
needed less time for the first for the Contains, Within, Covers and Covered by relations,
more time for the Touches, Instersects and Crosses relations, and approximately the
same time for the Disjoint relation.
This year, the SPIMBENCH track counted five participants: AML, Lily, LogMap,
FTRLIM and REMiner. REMiner participated for the first time this year while AML,
Lily, LogMap and FTRLIM also participated last year. The evaluation results of the
track are shown in Table 16. The results can also be found in HOBBIT git (https:
//hobbit-project.github.io/OAEI_2020.html).</p>
        </sec>
      </sec>
      <sec id="sec-5-2">
        <title>Sandbox Dataset ( 380 instances, 10000 triples)</title>
        <p>Lily and FTRLIM had the best performance overall both in terms of F-measure
and run time. Notably, their run time scaled very well with the increase in the
number of instances. REMiner produces the best results (almost full) for all metrics. Lily,
FTRLIM and AML had a higher recall than precision, while Lily and FTRLIM had a
full recall. By contrast, REMiner and LogMap had a higher precision and lower recall,
while REMiner had a full precision. AML, LogMap and REMiner had a similar run
time performance.
We evaluated all participants in the OAEI 2020. Unfortunately, none of the current
alignment systems can generate the coreferences between the cruise instances in the
Geolink Cruise benchmark. The state of the art alignment systems work well on finding
the links with a higher string similarity or string synonyms between two objects.
However, in terms of the instances with lower string similarities, or the external information
is not available or very limited to help the aligning task. Another kind of algorithm is
needed, like finding the relation of the instances based on the underlying structure of
the graphs. We hope that system will manage this track in future years.
We evaluated all SEALS participants in the OAEI (even those not registered for the
track) on a very small matching task33. This revealed that not all systems were able to
handle the task, and in the end, only the following systems were evaluated: ALOD2Vec,
AML, ATBox, DESKMatcher, LogMapKG, LogMapLt, Wiktionary. We also evaluated
LogMapBio but compared to LogMapKG it does not change the results (meaning that
the external knowledge does not help in these cases which is reasonable). LogMapKG
is the LogMap systems which returns TBox as well as ABox correspondences. In this
year, two systems registered especially for this track but were unable to finally submit
their system in time. This shows that there is a demand for this track and we plan
to provide this track also next year. We hope that the system developers are able to
submit the system next year. In comparison to the previous years, we have new matchers
like ALOD2Vec (which produced an error in 2018), ATBox (new), and DESKMatcher
(new).</p>
        <p>What did not change over the years is that some matchers do not return a valid
alignment file. The reason is the xml format of this file together with URIs in the
knowledge graph containing special characters e.g. ampersand. These characters should be
encoded, in order that xml parsers can process this file. Thus a post processing step is
executed which tries to create a valid xml file. The resulting alignments are available
for download. 34</p>
        <p>Table 17 shows the aggregated results for all systems, including the number of tasks
in which they were able to generate a non-empty alignment (#tasks) and the average
number of generated correspondences in those tasks (size). We report the macro
averaged precision, F-measure, and recall results where we do not distinguishing empty
and erroneous (or not generated) alignments. The values between parentheses show the
results when considering only non empty alignments.</p>
        <p>All systems were able to generate class correspondences. In terms of F-measure,
AML is still the best one and only DESKMatcher could not beat the baselines. The
recall values are higher than last year (maximum of 0.77) which shows that some matchers
improved and can find more class correspondences. Nevertheless there is still room for
improvement and some of these class matches looks like they are not easy to find.</p>
        <p>In the third year of this track all systems except the LogMap family are able to
return property correspondences. This is a huge improvement (which happens over the
years) because it makes the systems more usable in real case scenarios where a property
might not be classified as owl:ObjectProperty or owl:DatatypeProperty. The systems
ALOD2Vec, ATBox, and Wiktionary could achieve a F-measure of 0.95 or more which
shows that property matching is easier in this track than class or instance matching.</p>
        <p>With respect to instance correspondences, two systems (ALOD2Vec and
Wikitionary) exceed the best performance of last year with an F-measure of 0.87. The margin
between the baseline and the best systems is now a bit greater but still only 0.03 away.
Again LogMapKG returns a much higher number of instance correspondences (29,190
33 http://oaei.ontologymatching.org/2019/results/knowledgegraph/
small_test.zip
34 http://oaei.ontologymatching.org/2020/results/knowledgegraph/
oaei2020-knowledgegraph-alignments.zip</p>
        <p>ALOD2Vec 0:13:24
AML 0:50:55
ATBox 0:16:22
baselineAltLabel 0:10:57
baselineLabel 0:10:44
DESKMatcher 0:13:54
LogMapKG 2:47:51
LogMapLt 0:07:19
Wiktionary 0:30:12
ALOD2Vec 0:13:24
AML 0:50:55
ATBox 0:16:22
baselineAltLabel 0:10:57
baselineLabel 0:10:44
DESKMatcher 0:13:54
LogMapKG 2:47:51
LogMapLt 0:07:19
Wiktionary 0:30:12
ALOD2Vec 0:13:24
AML 0:50:55
ATBox 0:16:22
baselineAltLabel 0:10:57
baselineLabel 0:10:44
DESKMatcher 0:13:54
LogMapKG 2:47:51
LogMapLt 0:07:19
Wiktionary 0:30:12
ALOD2Vec 0:13:24
AML 0:50:55
ATBox 0:16:22
baselineAltLabel 0:10:57
baselineLabel 0:10:44
DESKMatcher 0:13:54
LogMapKG 2:47:51
LogMapLt 0:07:19
Wiktionary 0:30:12</p>
        <p>Class performance
5 20.0 1.00 0.80 0.67
5 23.6 0.98 0.89 0.81
5 25.6 0.97 0.87 0.79
5 16.4 1.00 0.74 0.59
5 16.4 1.00 0.74 0.59
5 91.4 0.76 0.71 0.66
5 24.0 0.95 0.84 0.76
4 23.0 0.80 (1.00) 0.56 (0.70) 0.43 (0.54)
5 22.4 1.00 0.80 0.67
in average) than all other participants but the recall is only slighly higher (0.03 to the
next best recall of 0.83).</p>
        <p>When analyzing the confidence values of the alignments, it turns out that most
matchers makes use of the range between zero and one. Only DESKMatcher,
LogMapLt, and the baselines return only 1.0. Further analysis can be made by browsing
to the dashboard 35 which is generated with the MELT framework [37].</p>
        <p>Regarding runtime, LogMapKG was was the slowest system (2:47:51 for all test
cases), followed by AML (0:50:55). Besides the baseline, four matchers were able to
compute the alignment in under 20 minutes which is a reasonable time for this track.</p>
        <p>In this year we also run the matchers in the hidden test cases to see how many
instance correspondences they return. The systems DESKMatcher, LogMapKG, and
AML (in test case starwars-lyrics) run into memory issues. Due to the fact that there is
no partial nor full gold standard available for these test cases, only the number of
returned instances correspondences is analyzed. In [35] we run the matchers from OAEI
2019 on these hidden test cases and manually evaluated 1,050 returned
correspondences. This results in the number of matches and a approximation of the precision
for each matcher and test case. Based on these values, the estimated number of true
positives for each test case can be calculated. The average and maximum number of
expected instance correspondences is shown in table 18 together with the number of
instance correspondences returned from OAEI 2020 matchers One can see that they
return 1-2 orders of magnitude more correspondences than the number of expected true
positives. Especially LogMapLt returns the highest number of correspondences in the
first two test cases and Wiktionary in the last test case. ATBox and AML return less
correspondences and a higher precision is expected in these test cases.
marvelcinematicuniverse 292.7 584.8
memoryalpha 73.6 285.5
starwars 48.5 109.1</p>
      </sec>
      <sec id="sec-5-3">
        <title>4.12 Interactive matching</title>
        <p>This year, three systems participated in the Interactive matching track. They are ALIN,
AML, and LogMap. Their results are shown in Table 19 and Figure 4 for both Anatomy
and Conference datasets.</p>
        <p>The table includes the following information (column names within parentheses):
– The performance of the system: Precision (Prec.), Recall (Rec.) and F-measure
(Fm.) with respect to the fixed reference alignment, as well as Recall+ (Rec.+) for the
35 http://oaei.ontologymatching.org/2020/results/knowledgegraph/
knowledge_graph_dashboard.html
0.874 0.456 0.599
0.915 0.705 0.796
0.75 0.679 0.713
0.612 0.648 0.629
0.516 0.617 0.562
0.841 0.659 0.739
0.91 0.698 0.79
0.843 0.682 0.754
0.777 0.677 0.723
0.721 0.65 0.684
–
–
–
–
–
–
–
–
–
–
–
–
–
–
–
NI stands for non-interactive, and refers to the results obtained by the matching system in the
original track.</p>
        <p>Anatomy task. To facilitate the assessment of the impact of user interactions, we
also provide the performance results from the original tracks, without interaction
(line with Error NI).
– To ascertain the impact of the oracle errors, we provide the performance of the
system with respect to the oracle (i.e., the reference alignment as modified by the
errors introduced by the oracle: Precision oracle (Prec. oracle), Recall oracle (Rec.
oracle) and F-measure oracle (F-m. oracle). For a perfect oracle these values match
the actual performance of the system.
– Total requests (Tot Reqs.) represents the number of distinct user interactions with
the tool, where each interaction can contain one to three conflicting
correspondences, that could be analysed simultaneously by a user.
– Distinct correspondences (Dist. Mapps) counts the total number of correspondences
for which the oracle gave feedback to the user (regardless of whether they were
submitted simultaneously, or separately).
– Finally, the performance of the oracle itself with respect to the errors it introduced
can be gauged through the positive precision (Pos. Prec.) and negative precision
(Neg. Prec.), which measure respectively the fraction of positive and negative
answers given by the oracle that are correct. For a perfect oracle these values are equal
to 1 (or 0, if no questions were asked).</p>
        <p>The figure shows the time intervals between the questions to the user/oracle for the
different systems and error rates. Different runs are depicted with different colors.</p>
        <p>The matching systems that participated in this track employ different
userinteraction strategies. While LogMap, and AML make use of user interactions
exclusively in the post-matching steps to filter their candidate correspondences, ALIN can
also add new candidate correspondences to its initial set. LogMap and AML both
request feedback on only selected correspondences candidates (based on their similarity
patterns or their involvement in unsatisfiabilities) and AML presents one
correspondence at a time to the user. ALIN and LogMap can both ask the oracle to analyze
several conflicting correspondences simultaneously.</p>
        <p>The performance of the systems usually improves when interacting with a perfect
oracle in comparison with no interaction. ALIN is the system that improves the most,
because its high number of oracle requests and its non-interactive performance was the
lowest of the interactive systems, and thus the easiest to improve.</p>
        <p>Although system performance deteriorates when the error rate increases, there are
still benefits from the user interaction—some of the systems’ measures stay above their
non-interactive values even for the larger error rates. Naturally, the more a system relies
on the oracle, the more its performance tends to be affected by the oracle’s errors.</p>
        <p>The impact of the oracle’s errors is linear for ALIN, and AML in most tasks, as
the F-measure according to the oracle remains approximately constant across all error
rates. It is supra-linear for LogMap in all datasets.</p>
        <p>Another aspect that was assessed, was the response time of systems, i.e., the time
between requests. Two models for system response times are frequently used in the
literature [16]: Shneiderman and Seow take different approaches to categorize the response
times taking a task-centered view and a user-centered view respectively. According to
task complexity, Shneiderman defines response time in four categories: typing, mouse
movement (50-150 ms), simple frequent tasks (1 s), common tasks (2-4 s) and complex
tasks (8-12 s). While Seow’s definition of response time is based on the user
expectations towards the execution of a task: instantaneous (100-200 ms), immediate (0.5-1
s), continuous (2-5 s), captive (7-10 s). Ontology alignment is a cognitively demanding
task and can fall into the third or fourth categories in both models. In this regard the
response times (request intervals as we call them above) observed in all datasets fall
into the tolerable and acceptable response times, and even into the first categories, in
both models. The request intervals for AML, LogMap and ALIN stay at a few
milliseconds for most datasets. It could be the case, however, that a user would not be able to
take advantage of these low response times because the task complexity may result in
higher user response time (i.e., the time the user needs to respond to the system after
the system is ready).</p>
        <p>Three systems were able to generate complex correspondences: AMLC, AROA, and
CANARD. The results for the other systems are reported in terms of simple alignments.
The results of the systems on the five test cases are summarized in Table 20.</p>
        <p>With respect to the Hydrography test cases, only AMLC can generate two correct
complex correspondences which are stating that a class in the source ontology is
equivalent to the union of two classes in the target ontology. Most of the systems achieved
fair results in terms of precision, but the low recall reflects that the current ontology
alignment systems still need to be improved to find more complex relations.</p>
        <p>In terms of Geolink and populated GeoLink test cases, the real-world instance data
from GeoLink Project is also populated into the ontology in order to enable the systems
that depend on instance-based matching algorithms to evaluate their performance. There
are three alignment systems that generate complex alignments in GeoLink Benchmark,
which are AMLC, AROA, and CANARD. AMLC didn’t find any correct complex
alignment, while AROA and CANARD achieved relatively good performance. One of
the reasons may be that these two systems are instance-based systems, which rely on the
shared instances between ontologies. In other words, the shared instance data between
two ontologies would be helpful to the matching process.</p>
        <p>In the populated Enslaved test case, only AMLC, AROA, and CANARD can
produce complex alignments. The relaxed precision of AMLC and AROA look relatively
fair, while CANARD reports a lower relaxed precision. AROA found the largest
number of the complex correspondences among three systems, while the AMLC outputs the
largest number of the simple correspondences.</p>
        <p>With respect to the Conference test cases the track has the same participant, AMLC,
as the last year. Based on the evaluation the alignments from AMLC now conforms to
the EDOAL syntax but otherwise the content of the alignment is the same.</p>
        <p>In the Populated Conference test case, AMLC’s results precision and coverage
scores are lower than last year, probably because it did not take a simple reference
alignment as input. CANARD’s results are close to last year’s. ALIN obtains the best
precision score.</p>
        <p>In the Taxon dataset, CANARD obtains the best coverage score but its precision has
decreased significantly. This year, AMLC could be evaluated on this dataset ; however,
the output correspondences did not cover the evaluation queries. The simple matcher
obtains approximatively the same coverage score.</p>
        <p>A more detailed discussion of the results of each task can be found in the OAEI page
for this track. For a third edition of complex matching in an OAEI campaign, and given
the inherent difficulty of the task, the results and participation are promising albeit still
modest.
5</p>
      </sec>
    </sec>
    <sec id="sec-6">
      <title>Conclusions and Lessons Learned</title>
      <p>In 2020, we witnessed a slight decrease in the number of participants in comparison
with previous years, but with a healthy mix of new and returning systems. However,
like last year, the distribution of participants by tracks was uneven. In future editions we
should facilitate the participation of non-Java systems (the use of the MELT framework
[36] was a step forward this year) and Machine Learning based system by providing
partial alignment sets for supervised learning. Furthermore, new systems might use
deep learning technology which requires specific hardware like GPUs and the like. An
option would be a simple HTTP interface to allow the deployment and evaluation on
different machines. The MELT framework can be easily extended with such an interface
while at the same time compatibility with SEALS and HOBBIT can be retained.</p>
      <p>The schema matching tracks saw abundant participation, but, as has been the trend
of the recent years, little substantial progress in terms of quality of the results or run
time of top matching systems, judging from the long-standing tracks. On the one hand,
this may be a sign of a performance plateau being reached by existing strategies and
algorithms, which would suggest that new technology is needed to obtain significant
improvements. On the other hand, it is also true that established matching systems tend
to focus more on new tracks and datasets than on improving their performance in
longstanding tracks, whereas new systems typically struggle to compete with established
ones.</p>
      <p>The number of matching systems capable of handling very large ontologies has
increased slightly over the last years, but is still relatively modest, judging from the Large
Biomedical Ontologies track. We will aim at facilitating participation in future editions
of this track by providing techniques to divide the matching tasks in manageable
subtasks (e.g., [40]).</p>
      <p>According to the Conference track there is still need for an improvement with
regard to the ability of matching systems to match properties. To assist system developers
in tackling this aspect we provided a more detailed evaluation in terms of the
analysis of the false positives per matching system (available on the Conference track web
page). This year this has been extended by the inspection of the explanation of the
correspondences provided by the systems. As already pointed out last year, less encouraging
is the low number of systems concerned with the logical coherence of the alignments
they produce, an aspect which is critical for several semantic web applications. Perhaps
a more direct approach is needed to promote this topic, such as providing a more
indepth analysis of the causes of incoherence in the evaluation or even organizing a future
track focusing on logical coherence alone. It is, however, clear that this is not an easy
task. When naively computing coherent alignments correct correspondences may be
removed and incorrect ones are kept, and therefore a domain expert should be involved
in the validation of different logical solutions [57, 49]. Finally, this year it was shown
that matching domain ontology to cross-domain ontology is difficult task for general
matching systems. While this has been done as an experiment without announcing
beforehand, we suppose to announce this as new test cases within the track for next year.</p>
      <p>With respect to the cross-lingual version of Conference, the MultiFarm track still
attracts a few number of participants implementing specific strategies to deal with
ontologies having a terminological layer in different natural languages. Despite this fact,
this year new participants came with alternative strategies (i.e, deep learning) with
respect to the last campaigns.</p>
      <p>The consensus-based evaluation in the Disease and Phenotype track offers limited
insights into performance, as several matching systems produce a number of unique
correspondences which may or may not be correct. In the absence of a true reference
alignment, future evaluation should seek to determine whether the unique
correspondences contain indicators of correctness, such as semantic similarity, or appear to be
noise. Comparison of the task results with embedded mappings of equivalence in the
MONDO disease ontology can also be investigated in future evaluation [55].</p>
      <p>Despite the quite promising results obtained by matching systems for the
Biodiversity and Ecology track, the most important observation is that none of the systems has
been able to detect mappings established by domain experts. Detecting such
correspondences requires the use of domain-specific core knowledge that captures biodiversity
concepts. In addition this year, we put the light on the quasi total incapacity of systems
to handle SKOS as input format for semantic resources to align.</p>
      <p>The interactive matching track also witnessed a small number of participants.
Three systems participated this year. This is puzzling considering that this track is based
on the Anatomy and Conference test cases, and those tracks had 13 participants. The
process of programmatically querying the Oracle class used to simulate user
interactions is simple enough that it should not be a deterrent for participation, but perhaps
we should look at facilitating the process further in future OAEI editions by providing
implementation examples.</p>
      <p>The complex matching track opens new perspectives in the field of ontology
matching. Tackling complex matching automatically is extremely challenging, likely
requiring profound adaptations from matching systems, so the fact that there were three
participants that were able to generate complex correspondences in this track should
be seen as a positive sign of progress to the state of the art in ontology matching. This
year automatic evaluation has been introduced following an instance-based comparison
approach.</p>
      <p>In the instance matching tracks participation increased this year for SPIMBENCH
as systems became more familiar with the HOBBIT platform and had more time to do
the migration. Regarding Spatial benchmark, the systems didn’t have newer versions
and the number of participants remained the same. Thus, the benchmark and the systems
were the exact same as last year. Participation might increase next year as the systems
are still updating their versions and new systems are under development. Automatic
instance-matching benchmark generation algorithms have been gaining popularity, as
evidenced by the fact that they are used in all three instance matching tracks of this
OAEI edition. One aspect that has not been addressed in such algorithms is that, if the
transformation is too extreme, the correspondence may be unrealistic and impossible to
detect even by humans. As such, we argue that human-in-the-loop techniques can be
exploited to do a preventive quality-checking of generated correspondences, and refine
the set of correspondences included in the final reference alignment.</p>
      <p>In the knowledge graph track, more matchers are able to match rdf:Properties and
are thus better suited for real matching cases. In the third year of this track we saw a
small improvement in instance alignments but the margin to the baselines is still small.
In this year two new systems focused on the KG track but could not submit their systems
in time. We thus expect more systems in the upcoming year.</p>
      <p>Like in previous OAEI editions, most participants provided a description of their
systems and their experience in the evaluation, in the form of OAEI system papers.
These papers, like the present one, have not been peer reviewed. However, they are full
contributions to this evaluation exercise, reflecting the effort and insight of matching
systems developers, and providing details about those systems and the algorithms they
implement.</p>
      <p>As each year, fruitful discussions at the Ontology Matching point out different
directions for future improvements in OAEI. In particular, in terms of new use cases, one
potential new track involves matching ontologies of units of measure (OM and QUDT)
[51], in order to improve the ability of a digital twin platform to harmonise, integrate and
process quantity values. Another track to be included in the next campaign is about the
chemical/biological laboratory domain with strong interest from pharmaceutical
companies [30, 32].</p>
      <p>The Ontology Alignment Evaluation Initiative will strive to remain a reference to
the ontology matching community by improving both the test cases and the testing
methodology to better reflect actual needs, as well as to promote progress in this field.
More information can be found at: http://oaei.ontologymatching.org.</p>
    </sec>
    <sec id="sec-7">
      <title>Acknowledgements</title>
      <p>We warmly thank the participants of this campaign. We know that they have worked
hard to have their matching tools executable in time and they provided useful reports
on their experience. The best way to learn about the results remains to read the papers
that follow.</p>
      <p>We are also grateful to Martin Ringwald and Terry Hayamizu for providing the
reference alignment for the anatomy ontologies and thank Elena Beisswanger for her
thorough support on improving the quality of the dataset.</p>
      <p>We thank Andrea Turbati and the AGROVOC team for their very appreciated help
with the preparation of the AGROVOC subset ontology. We are also grateful to
Catherine Roussey and Nathalie Hernandez for their help on the Taxon alignment.</p>
      <p>We also thank for their support the past members of the Ontology Alignment
Evaluation Initiative steering committee: Je´roˆme Euzenat (INRIA, FR), Yannis Kalfoglou
(Ricoh laboratories, UK), Miklos Nagy (The Open University,UK), Natasha Noy
(Google Inc., USA), Yuzhong Qu (Southeast University, CN), York Sure
(Leibniz Gemeinschaft, DE), Jie Tang (Tsinghua University, CN), Heiner Stuckenschmidt
(Mannheim Universita¨t, DE), George Vouros (University of the Aegean, GR).</p>
      <p>Daniel Faria was supported by the EC H2020 grant 676559
ELIXIREXCELERATE and the Portuguese FCT Grant 22231 BioData.pt, co-financed by
FEDER.</p>
      <p>Ernesto Jimenez-Ruiz has been partially supported by the SIRIUS Centre for
Scalable Data Access (Research Council of Norway, project no.: 237889) and the AIDA
project (Alan Turing Institute).</p>
      <p>Catia Pesquita was supported by the FCT through the LASIGE Strategic Project
(UID/CEC/00408/2013) and the research grant PTDC/EEI-ESS/4633/2014.</p>
      <p>Irini Fundulaki and Tzanina Saveta were supported by the EU’s Horizon 2020
research and innovation programme under grant agreement No 688227 (Hobbit).</p>
      <p>Jana Vatasˇcˇinova´ and Ondrˇej Zamazal were supported by the CSF grant no.
1823964S.</p>
      <p>Patrick Lambrix, Huanyu Li, Mina Abd Nikooie Pour and Ying Li have been
supported by the Swedish e-Science Research Centre (SeRC), the Swedish Research
Council (Vetenskapsra˚det, dnr 2018-04147) and the Swedish National Graduate School in
Computer Science (CUGS).</p>
      <p>Lu Zhou and Pascal Hitzler have been supported by the National Science
Foundation under Grant No. 2033521, KnowWhereGraph: Enriching and Linking
CrossDomain Knowledge Graphs using Spatially-Explicit AI Technologies and the Andrew
W. Mellon Foundation through the Enslaved project (identifiers 1708-04732 and
190206575).</p>
      <p>Beyza Yaman has been supported by the European Union’s Horizon 2020
research and innovation programme under Marie Sklodowska-Curie grant agreement No.
801522, by Science Foundation Ireland and co-funded by the European Regional
Development Fund through the ADAPT Centre for Digital Content Technology [grant
number 13/RC/2106] and Ordnance Survey Ireland.</p>
      <p>The Biodiversity and Ecology track has been partially funded by the German
Research Foundation in the context of the GFBio Project (grant No. SE 553/7-1) and the
CRC 1076 AquaDiva, the Leitprojekt der Fraunhofer Gesellschaft in the context of the
MED2ICIN project (grant No. 600628) and the German Network for Bioinformatics
Infrastructure - de.NBI (grant No. 031A539B). In 2020, the track was also supported
by the Data to Knowledge in Agronomy and Biodiversity (D2KAB – www.d2kab.org)
project that received funding from the French National Research Agency
(ANR-18CE23-0017). We would like to thank FAO AIMS and US NAL as well as the GACS
project for providing mappings betweem AGROVOC and NALT. We would like to
thank Christian Pichot and the ANAEE France project for providing mappings betweem
ANAEETHES and GEMET.
5. Alsayed Algergawy, Daniel Faria, Alfio Ferrara, Irini Fundulaki, Ian Harrow, Sven Hertling,
Ernesto Jime´nez-Ruiz, Naouel Karam, Abderrahmane Khiat, Patrick Lambrix, Huanyu Li,
Stefano Montanelli, Heiko Paulheim, Catia Pesquita, Tzanina Saveta, Pavel Shvaiko,
Andrea Splendiani, E´lodie Thie´blin, Ca´ssia Trojahn, Jana Vatascinova´, Ondrej Zamazal, and
Lu Zhou. Results of the ontology alignment evaluation initiative 2019. In Proceedings of the
14th International Workshop on Ontology Matching, Auckland, New Zealand, pages 46–85,
2019.
6. R Amini, L Zhou, and P Hitzler. Geolink cruises: A non-synthetic benchmark for
coreference resolution on knowledge graphs. In 29th ACM International Conference on
Information and Knowledge Management, 2020.
7. Benhamin Ashpole, Marc Ehrig, Je´r oˆme Euzenat, and Heiner Stuckenschmidt, editors. Proc.</p>
      <p>
        K-Cap Workshop on Integrating Ontologies, Banff (Canada), 2005.
8. Olivier Bodenreider. The unified medical language system (UMLS): integrating biomedical
terminology. Nucleic Acids Research, 32:267–270, 2004.
9. Pier Luigi Buttigieg, Norman Morrison, Barry Smith, Christopher J. Mungall, and
Suzanna E. Lewis. The environment ontology: contextualising biological and biomedical
entities. Biomedical Semantics, 4(
        <xref ref-type="bibr" rid="ref1">1</xref>
        ):43, December 2013.
10. Caterina Caracciolo, Je´roˆ me Euzenat, Laura Hollink, Ryutaro Ichise, Antoine Isaac,
Ve´ronique Malaise´, Christian Meilicke, Juan Pane, Pavel Shvaiko, Heiner Stuckenschmidt,
Ondrej Sva´b-Zamazal, and Vojtech Sva´tek. Results of the ontology alignment evaluation
initiative 2008. In Proceedings of the 3rd Ontology matching workshop, Karlsruhe (DE),
pages 73–120, 2008.
11. Michelle Cheatham, Zlatan Dragisic, Je´roˆ me Euzenat, Daniel Faria, Alfio Ferrara, Giorgos
Flouris, Irini Fundulaki, Roger Granada, Valentina Ivanova, Ernesto Jime´nez-Ruiz, Patrick
Lambrix, Stefano Montanelli, Catia Pesquita, Tzanina Saveta, Pavel Shvaiko, Alessandro
Solimando, Ca´ssia Trojahn, and Ondrˇej Zamazal. Results of the ontology alignment
evaluation initiative 2015. In Proceedings of the 10th International Ontology matching workshop,
Bethlehem (PA, US), pages 60–115, 2015.
12. Michelle Cheatham, Dalia Varanka, Fatima Arauz, and Lu Zhou. Alignment of surface water
ontologies: a comparison of manual and automated approaches. J. Geogr. Syst., 22(
        <xref ref-type="bibr" rid="ref2">2</xref>
        ):267–
289, 2020.
13. Jean Clobert, Andre´ Chanzy, Jean-Franc¸ois Le Galliard, Abad Chabbi, Lucile Greiveldinger,
Thierry Caquet, Michel Loreau, Christian Mougin, Christian Pichot, Jacques Roy, et al.
How to integrate experimental research approaches in ecological and environmental
studies: Anaee france as an example. Frontiers in Ecology and Evolution, 6:43, 2018.
14. Laurel Cooper, Ramona L. Walls, Justin Elser, Maria A. Gandolfo, Dennis W. Stevenson,
Barry Smith, Justin Preece, Balaji Athreya, Christopher J. Mungall, Stefan Rensing, Manuel
Hiss, Daniel Lang, Ralf Reski, Tanya Z. Berardini, Donghui Li, Eva Huala, Mary
Schaeffer, Naama Menda, Elizabeth Arnaud, Rosemary Shrestha, Yukiko Yamazaki, and Pankaj
Jaiswal. The Plant Ontology as a Tool for Comparative Plant Anatomy and Genomic
Analyses. Plant and Cell Physiology, 54(
        <xref ref-type="bibr" rid="ref2">2</xref>
        ):e1, December 2012.
15. Bernardo Cuenca Grau, Zlatan Dragisic, Kai Eckert, Je´roˆ me Euzenat, Alfio Ferrara, Roger
Granada, Valentina Ivanova, Ernesto Jime´nez-Ruiz, Andreas Oskar Kempf, Patrick
Lambrix, Andriy Nikolov, Heiko Paulheim, Dominique Ritze, Franc¸ois Scharffe, Pavel Shvaiko,
Ca´ssia Trojahn dos Santos, and Ondrej Zamazal. Results of the ontology alignment
evaluation initiative 2013. In Pavel Shvaiko, Je´roˆ me Euzenat, Kavitha Srinivas, Ming Mao,
and Ernesto Jime´nez-Ruiz, editors, Proceedings of the 8th International Ontology matching
workshop, Sydney (NSW, AU), pages 61–100, 2013.
16. Jim Dabrowski and Ethan V. Munson. 40 years of searching for the best computer system
response time. Interacting with Computers, 23(5):555–564, 2011.
61. Martin Sˇ atra and Ondrej Zamazal. Towards matching of domain ontologies to cross-domain
ontology: Evaluation perspective. In Proceedings of the 19th International Workshop on
Ontology Matching, 2020.
62. Tzanina Saveta, Evangelia Daskalaki, Giorgos Flouris, Irini Fundulaki, Melanie Herschel,
and Axel-Cyrille Ngonga Ngomo. Pushing the limits of instance matching systems: A
semantics-aware benchmark for linked data. In Proceedings of the 24th International
Conference on World Wide Web, pages 105–106, New York, NY, USA, 2015. ACM.
63. Alessandro Solimando, Ernesto Jime´nez-Ruiz, and Giovanna Guerrini. Detecting and
correcting conservativity principle violations in ontology-to-ontology mappings. In Proceedings
of the International Semantic Web Conference, pages 1–16. Springer, 2014.
64. Alessandro Solimando, Ernesto Jimenez-Ruiz, and Giovanna Guerrini. Minimizing
conservativity violations in ontology alignments: Algorithms and evaluation. Knowledge and
Information Systems, 2016.
65. Christian Strobl. Encyclopedia of GIS, chapter Dimensionally Extended Nine-Intersection
      </p>
      <p>Model (DE-9IM), pages 240–245. Springer, 2008.
66. York Sure, Oscar Corcho, Je´roˆ me Euzenat, and Todd Hughes, editors. Proceedings of the</p>
      <p>
        Workshop on Evaluation of Ontology-based Tools (EON), Hiroshima (JP), 2004.
67. E´lodie Thie´blin. Do competency questions for alignment help fostering complex
correspondences? In Proceedings of the EKAW Doctoral Consortium 2018, 2018.
68. E´lodie Thie´blin, Fabien Amarger, Ollivier Haemmerle´, Nathalie Hernandez, and Ca´ssia
Trojahn dos Santos. Rewriting SELECT SPARQL queries from 1: n complex correspondences.
In Proceedings of the 11th International Workshop on Ontology Matching, pages 49–60,
2016.
69. Elodie Thie´blin, Michelle Cheatham, Cassia Trojahn, Ondrej Zamazal, and Lu Zhou. The
First Version of the OAEI Complex Alignment Benchmark. In Proceedings of the
International Semantic Web Conference (Posters and Demos), 2018.
70. Ondrˇej Zamazal and Vojteˇch Sva´tek. The ten-year ontofarm and its fertilization within the
onto-sphere. Web Semantics: Science, Services and Agents on the World Wide Web, 43:46–
53, 2017.
71. L Zhou, C Shimizu, P Hitzler, A Sheill, S Estrecha, C Foley, D Tarr, and Rehberger D. The
enslaved dataset: A real-world complex ontology alignment benchmark using wikibase. In
29th ACM International Conference on Information and Knowledge Management, 2020.
72. Lu Zhou, Michelle Cheatham, Adila Krisnadhi, and Pascal Hitzler. A complex alignment
benchmark: Geolink dataset. In Proceedings of the 17th International Semantic Web
Conference, Monterey (CA, USA), pages 273–288, 2018.
73. Lu Zhou, Michelle Cheatham, Adila Krisnadhi, and Pascal Hitzler. Geolink data set: A
complex alignment benchmark from real-world ontology. Data Intell., 2(
        <xref ref-type="bibr" rid="ref3">3</xref>
        ):353–378, 2020.
      </p>
    </sec>
  </body>
  <back>
    <ref-list>
      <ref id="ref1">
        <mixed-citation>
          1.
          <string-name>
            <given-names>Manel</given-names>
            <surname>Achichi</surname>
          </string-name>
          , Michelle Cheatham, Zlatan Dragisic, Je´roˆme Euzenat, Daniel Faria, Alfio Ferrara, Giorgos Flouris, Irini Fundulaki, Ian Harrow, Valentina Ivanova, Ernesto Jime´nezRuiz,
          <string-name>
            <surname>Kristian</surname>
            <given-names>Kolthoff</given-names>
          </string-name>
          , Elena Kuss, Patrick Lambrix, Henrik Leopold,
          <string-name>
            <given-names>Huanyu</given-names>
            <surname>Li</surname>
          </string-name>
          ,
          <string-name>
            <given-names>Christian</given-names>
            <surname>Meilicke</surname>
          </string-name>
          , Majid Mohammadi, Stefano Montanelli, Catia Pesquita, Tzanina Saveta, Pavel Shvaiko, Andrea Splendiani,
          <string-name>
            <given-names>Heiner</given-names>
            <surname>Stuckenschmidt</surname>
          </string-name>
          , E´lodie Thie´blin, Konstantin Todorov, Ca´ssia Trojahn, and
          <string-name>
            <given-names>Ondrej</given-names>
            <surname>Zamazal</surname>
          </string-name>
          .
          <article-title>Results of the ontology alignment evaluation initiative 2017</article-title>
          .
          <source>In Proceedings of the 12th International Workshop on Ontology Matching</source>
          , Vienna, Austria, pages
          <fpage>61</fpage>
          -
          <lpage>113</lpage>
          ,
          <year>2017</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref2">
        <mixed-citation>
          2.
          <string-name>
            <given-names>Manel</given-names>
            <surname>Achichi</surname>
          </string-name>
          , Michelle Cheatham, Zlatan Dragisic, Jerome Euzenat, Daniel Faria, Alfio Ferrara, Giorgos Flouris, Irini Fundulaki, Ian Harrow, Valentina Ivanova, Ernesto Jime´nezRuiz,
          <string-name>
            <surname>Elena</surname>
            <given-names>Kuss</given-names>
          </string-name>
          , Patrick Lambrix, Henrik Leopold,
          <string-name>
            <given-names>Huanyu</given-names>
            <surname>Li</surname>
          </string-name>
          ,
          <string-name>
            <given-names>Christian</given-names>
            <surname>Meilicke</surname>
          </string-name>
          , Stefano Montanelli, Catia Pesquita, Tzanina Saveta, Pavel Shvaiko, Andrea Splendiani, Heiner Stuckenschmidt, Konstantin Todorov, Ca´ssia Trojahn, and
          <string-name>
            <given-names>Ondrej</given-names>
            <surname>Zamazal</surname>
          </string-name>
          .
          <article-title>Results of the ontology alignment evaluation initiative 2016</article-title>
          .
          <source>In Proceedings of the 11th International Ontology matching workshop</source>
          ,
          <source>Kobe (JP)</source>
          , pages
          <fpage>73</fpage>
          -
          <lpage>129</lpage>
          ,
          <year>2016</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref3">
        <mixed-citation>
          3. Jose´ Luis Aguirre, Bernardo Cuenca Grau, Kai Eckert, Je´roˆme Euzenat, Alfio Ferrara, Robert Willem van Hague,
          <string-name>
            <surname>Laura Hollink</surname>
          </string-name>
          , Ernesto Jime´
          <article-title>nez-</article-title>
          <string-name>
            <surname>Ruiz</surname>
            ,
            <given-names>Christian</given-names>
          </string-name>
          <string-name>
            <surname>Meilicke</surname>
          </string-name>
          , Andriy Nikolov, Dominique Ritze, Franc¸ois Scharffe, Pavel Shvaiko, Ondrej Sva´
          <fpage>b</fpage>
          -Zamazal, Ca´ssia Trojahn, and
          <string-name>
            <given-names>Benjamin</given-names>
            <surname>Zapilko</surname>
          </string-name>
          .
          <article-title>Results of the ontology alignment evaluation initiative 2012</article-title>
          .
          <source>In Proceedings of the 7th International Ontology matching workshop</source>
          , Boston (MA, US), pages
          <fpage>73</fpage>
          -
          <lpage>115</lpage>
          ,
          <year>2012</year>
          .
        </mixed-citation>
      </ref>
      <ref id="ref4">
        <mixed-citation>
          4.
          <string-name>
            <given-names>Alsayed</given-names>
            <surname>Algergawy</surname>
          </string-name>
          , Michelle Cheatham, Daniel Faria, Alfio Ferrara, Irini Fundulaki, Ian Harrow, Sven Hertling, Ernesto Jime´
          <fpage>nez</fpage>
          -Ruiz, Naouel Karam, Abderrahmane Khiat, Patrick Lambrix,
          <string-name>
            <given-names>Huanyu</given-names>
            <surname>Li</surname>
          </string-name>
          ,
          <string-name>
            <given-names>Stefano</given-names>
            <surname>Montanelli</surname>
          </string-name>
          , Heiko Paulheim, Catia Pesquita, Tzanina Saveta, Daniela Schmidt, Pavel Shvaiko,
          <string-name>
            <given-names>Andrea</given-names>
            <surname>Splendiani</surname>
          </string-name>
          , E´ lodie Thie´blin, Ca´ssia Trojahn, Jana Vatascinova´,
          <string-name>
            <given-names>Ondrej</given-names>
            <surname>Zamazal</surname>
          </string-name>
          , and
          <string-name>
            <given-names>Lu</given-names>
            <surname>Zhou</surname>
          </string-name>
          .
          <article-title>Results of the ontology alignment evaluation initiative 2018</article-title>
          .
          <source>In Proceedings of the 13th International Workshop on Ontology Matching</source>
          , Monterey (CA, US), pages
          <fpage>76</fpage>
          -
          <lpage>116</lpage>
          ,
          <year>2018</year>
          .
        </mixed-citation>
      </ref>
    </ref-list>
  </back>
</article>